{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,14]],"date-time":"2026-04-14T22:57:57Z","timestamp":1776207477451,"version":"3.50.1"},"reference-count":54,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100002703","name":"Jiangsu University","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100002703","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computer Vision and Image Understanding"],"published-print":{"date-parts":[[2026,4]]},"DOI":"10.1016\/j.cviu.2026.104741","type":"journal-article","created":{"date-parts":[[2026,3,18]],"date-time":"2026-03-18T10:16:05Z","timestamp":1773828965000},"page":"104741","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["CLIP-driven fine-grained mining for text-based person search"],"prefix":"10.1016","volume":"267","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-9718-6660","authenticated-orcid":false,"given":"Xianwen","family":"Lin","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7073-0813","authenticated-orcid":false,"given":"Xia","family":"Geng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shengli","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhi","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.cviu.2026.104741_b1","doi-asserted-by":"crossref","unstructured":"Aggarwal,\u00a0S., Radhakrishnan,\u00a0V.B., Chakraborty,\u00a0A., 2020. Text-based person search via attribute-aided matching. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision. pp. 2617\u20132625.","DOI":"10.1109\/WACV45572.2020.9093640"},{"key":"10.1016\/j.cviu.2026.104741_b2","series-title":"Rasa: Relation and sensitivity aware representation learning for text-based person search","author":"Bai","year":"2023"},{"key":"10.1016\/j.cviu.2026.104741_b3","series-title":"European Conference on Computer Vision","first-page":"213","article-title":"End-to-end object detection with transformers","author":"Carion","year":"2020"},{"key":"10.1016\/j.cviu.2026.104741_b4","series-title":"European Conference on Computer Vision","first-page":"104","article-title":"Uniter: Universal image-text representation learning","author":"Chen","year":"2020"},{"key":"10.1016\/j.cviu.2026.104741_b5","doi-asserted-by":"crossref","first-page":"171","DOI":"10.1016\/j.neucom.2022.04.081","article-title":"TIPCB: A simple but effective part-based convolutional baseline for text-based person search","volume":"494","author":"Chen","year":"2022","journal-title":"Neurocomputing"},{"key":"10.1016\/j.cviu.2026.104741_b6","series-title":"Masked-attention mask transformer for universal image segmentation","author":"Cheng","year":"2021"},{"issue":"11","key":"10.1016\/j.cviu.2026.104741_b7","doi-asserted-by":"crossref","first-page":"12994","DOI":"10.1109\/TII.2024.3431044","article-title":"EF-DETR: A lightweight transformer-based object detector with an encoder-free neck","volume":"20","author":"Cheng","year":"2024","journal-title":"IEEE Trans. Ind. Informat."},{"key":"10.1016\/j.cviu.2026.104741_b8","series-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2018"},{"key":"10.1016\/j.cviu.2026.104741_b9","series-title":"Semantically self-aligned network for text-to-image part-aware person re-identification","author":"Ding","year":"2021"},{"key":"10.1016\/j.cviu.2026.104741_b10","series-title":"An image is worth 16x16 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2020"},{"key":"10.1016\/j.cviu.2026.104741_b11","first-page":"4477","article-title":"AXM-net: Implicit cross-modal feature alignment for person re-identification","volume":"vol. 36","author":"Farooq","year":"2022"},{"key":"10.1016\/j.cviu.2026.104741_b12","doi-asserted-by":"crossref","unstructured":"Fujii,\u00a0T., Tarashima,\u00a0S., 2023. Bilma: Bidirectional local-matching for text-based person re-identification. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 2786\u20132790.","DOI":"10.1109\/ICCVW60793.2023.00295"},{"key":"10.1016\/j.cviu.2026.104741_b13","series-title":"Contextual non-local alignment over full-scale representation for text-based person search","author":"Gao","year":"2021"},{"key":"10.1016\/j.cviu.2026.104741_b14","series-title":"What do vision transformers learn? a visual exploration","author":"Ghiasi","year":"2022"},{"key":"10.1016\/j.cviu.2026.104741_b15","series-title":"Text-based person search with limited data","author":"Han","year":"2021"},{"key":"10.1016\/j.cviu.2026.104741_b16","doi-asserted-by":"crossref","first-page":"163","DOI":"10.1109\/TIP.2023.3337653","article-title":"VGSG: Vision-guided semantic-group network for text-based person search","volume":"33","author":"He","year":"2023","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.cviu.2026.104741_b17","doi-asserted-by":"crossref","unstructured":"Hu,\u00a0J., Shen,\u00a0L., Sun,\u00a0G., 2018. Squeeze-and-excitation networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 7132\u20137141.","DOI":"10.1109\/CVPR.2018.00745"},{"key":"10.1016\/j.cviu.2026.104741_b18","doi-asserted-by":"crossref","unstructured":"Jiang,\u00a0D., Ye,\u00a0M., 2023. Cross-modal implicit relation reasoning and aligning for text-to-image person retrieval. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 2787\u20132797.","DOI":"10.1109\/CVPR52729.2023.00273"},{"key":"10.1016\/j.cviu.2026.104741_b19","doi-asserted-by":"crossref","first-page":"35631","DOI":"10.52202\/075280-1549","article-title":"Learning mask-aware clip representations for zero-shot segmentation","volume":"36","author":"Jiao","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104741_b20","first-page":"4138","article-title":"Pedestrian attribute recognition: A new benchmark dataset and a large language model augmented framework","volume":"vol. 39","author":"Jin","year":"2025"},{"key":"10.1016\/j.cviu.2026.104741_b21","first-page":"11189","article-title":"Pose-guided multi-granularity attention network for text-based person search","volume":"vol. 34","author":"Jing","year":"2020"},{"key":"10.1016\/j.cviu.2026.104741_b22","doi-asserted-by":"crossref","unstructured":"Kirillov,\u00a0A., Mintun,\u00a0E., Ravi,\u00a0N., Mao,\u00a0H., Rolland,\u00a0C., Gustafson,\u00a0L., Xiao,\u00a0T., Whitehead,\u00a0S., Berg,\u00a0A.C., Lo,\u00a0W.-Y., et al., 2023. Segment anything. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 4015\u20134026.","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"10.1016\/j.cviu.2026.104741_b23","series-title":"International Conference on Machine Learning","first-page":"12888","article-title":"Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation","author":"Li","year":"2022"},{"key":"10.1016\/j.cviu.2026.104741_b24","first-page":"9694","article-title":"Align before fuse: Vision and language representation learning with momentum distillation","volume":"34","author":"Li","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104741_b25","doi-asserted-by":"crossref","unstructured":"Li,\u00a0S., Xiao,\u00a0T., Li,\u00a0H., Zhou,\u00a0B., Yue,\u00a0D., Wang,\u00a0X., 2017. Person search with natural language description. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 1970\u20131979.","DOI":"10.1109\/CVPR.2017.551"},{"issue":"3","key":"10.1016\/j.cviu.2026.104741_b26","doi-asserted-by":"crossref","first-page":"1624","DOI":"10.1109\/TCSVT.2021.3073718","article-title":"Transformer-based language-person search with multiple region slicing","volume":"32","author":"Li","year":"2021","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.cviu.2026.104741_b27","first-page":"5703","article-title":"Dm-adapter: Domain-aware mixture-of-adapters for text-based person retrieval","volume":"vol. 39","author":"Liu","year":"2025"},{"key":"10.1016\/j.cviu.2026.104741_b28","article-title":"Multi-modal reference learning for fine-grained text-to-image retrieval","author":"Ma","year":"2025","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.cviu.2026.104741_b29","doi-asserted-by":"crossref","unstructured":"Ma,\u00a0Y., Sun,\u00a0X., Ji,\u00a0J., Jiang,\u00a0G., Zhuang,\u00a0W., Ji,\u00a0R., 2023. Beat: Bi-directional One-to-Many Embedding Alignment for Text-based Person Retrieval. In: Proceedings of the 31st ACM International Conference on Multimedia. pp. 4157\u20134168.","DOI":"10.1145\/3581783.3611768"},{"key":"10.1016\/j.cviu.2026.104741_b30","series-title":"International Conference on Machine Learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.cviu.2026.104741_b31","doi-asserted-by":"crossref","unstructured":"Sarafianos,\u00a0N., Xu,\u00a0X., Kakadiaris,\u00a0I.A., 2019. Adversarial representation learning for text-to-image matching. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 5814\u20135824.","DOI":"10.1109\/ICCV.2019.00591"},{"key":"10.1016\/j.cviu.2026.104741_b32","doi-asserted-by":"crossref","unstructured":"Shao,\u00a0Z., Zhang,\u00a0X., Ding,\u00a0C., Wang,\u00a0J., Wang,\u00a0J., 2023. Unified pre-training with pseudo texts for text-to-image person re-identification. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 11174\u201311184.","DOI":"10.1109\/ICCV51070.2023.01026"},{"key":"10.1016\/j.cviu.2026.104741_b33","doi-asserted-by":"crossref","unstructured":"Shao,\u00a0Z., Zhang,\u00a0X., Fang,\u00a0M., Lin,\u00a0Z., Wang,\u00a0J., Ding,\u00a0C., 2022. Learning Granularity-Unified Representations for Text-to-Image Person Re-identification. In: Proceedings of the 30th ACM International Conference on Multimedia. pp. 5566\u20135574.","DOI":"10.1145\/3503161.3548028"},{"key":"10.1016\/j.cviu.2026.104741_b34","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2024.112893","article-title":"Enhancing visual representation for text-based person searching","volume":"309","author":"Shen","year":"2025","journal-title":"Knowl.-Based Syst."},{"issue":"7","key":"10.1016\/j.cviu.2026.104741_b35","doi-asserted-by":"crossref","first-page":"4390","DOI":"10.1109\/TCSVT.2021.3128214","article-title":"Large-scale spatio-temporal person re-identification: Algorithms and benchmark","volume":"32","author":"Shu","year":"2021","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.cviu.2026.104741_b36","series-title":"See finer, see more: Implicit modality alignment for text-based person retrieval","author":"Shu","year":"2022"},{"key":"10.1016\/j.cviu.2026.104741_b37","series-title":"Vl-bert: Pre-training of generic visual-linguistic representations","author":"Su","year":"2019"},{"key":"10.1016\/j.cviu.2026.104741_b38","doi-asserted-by":"crossref","unstructured":"Sun,\u00a0Y., Zheng,\u00a0L., Yang,\u00a0Y., Tian,\u00a0Q., Wang,\u00a0S., 2018. Beyond part models: Person retrieval with refined part pooling (and a strong convolutional baseline). In: Proceedings of the European Conference on Computer Vision. ECCV, pp. 480\u2013496.","DOI":"10.1007\/978-3-030-01225-0_30"},{"key":"10.1016\/j.cviu.2026.104741_b39","series-title":"Graph attention networks","author":"Veli\u010dkovi\u0107","year":"2017"},{"key":"10.1016\/j.cviu.2026.104741_b40","series-title":"European Conference on Computer Vision","first-page":"402","article-title":"Vitaa: Visual-textual attributes alignment in person search by natural language","author":"Wang","year":"2020"},{"key":"10.1016\/j.cviu.2026.104741_b41","article-title":"Pedestrian attribute recognition via clip based prompt vision-language fusion","author":"Wang","year":"2024","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.cviu.2026.104741_b42","doi-asserted-by":"crossref","unstructured":"Wang,\u00a0Z., Zhu,\u00a0A., Xue,\u00a0J., Wan,\u00a0X., Liu,\u00a0C., Wang,\u00a0T., Li,\u00a0Y., 2022a. Caibc: Capturing all-round information beyond color for text-based person retrieval. In: Proceedings of the 30th ACM International Conference on Multimedia. pp. 5314\u20135322.","DOI":"10.1145\/3503161.3548057"},{"key":"10.1016\/j.cviu.2026.104741_b43","doi-asserted-by":"crossref","unstructured":"Wang,\u00a0Z., Zhu,\u00a0A., Xue,\u00a0J., Wan,\u00a0X., Liu,\u00a0C., Wang,\u00a0T., Li,\u00a0Y., 2022b. Look before you leap: Improving text-based person retrieval by learning a consistent cross-modal common manifold. In: Proceedings of the 30th ACM International Conference on Multimedia. pp. 1984\u20131992.","DOI":"10.1145\/3503161.3548166"},{"key":"10.1016\/j.cviu.2026.104741_b44","doi-asserted-by":"crossref","unstructured":"Xu,\u00a0M., Zhang,\u00a0Z., Wei,\u00a0F., Hu,\u00a0H., Bai,\u00a0X., 2023. Side adapter network for open-vocabulary semantic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 2945\u20132954.","DOI":"10.1109\/CVPR52729.2023.00288"},{"key":"10.1016\/j.cviu.2026.104741_b45","doi-asserted-by":"crossref","unstructured":"Yan,\u00a0S., Dong,\u00a0N., Liu,\u00a0J., Zhang,\u00a0L., Tang,\u00a0J., 2023. Learning comprehensive representations with richer self for text-to-image person re-identification. In: Proceedings of the 31st ACM International Conference on Multimedia. pp. 6202\u20136211.","DOI":"10.1145\/3581783.3611832"},{"key":"10.1016\/j.cviu.2026.104741_b46","series-title":"CLIP-driven fine-grained text-image person re-identification","author":"Yan","year":"2022"},{"key":"10.1016\/j.cviu.2026.104741_b47","doi-asserted-by":"crossref","unstructured":"Yang,\u00a0S., Zhou,\u00a0Y., Zheng,\u00a0Z., Wang,\u00a0Y., Zhu,\u00a0L., Wu,\u00a0Y., 2023. Towards unified text-based person retrieval: A large-scale multi-attribute and language search benchmark. In: Proceedings of the 31st ACM International Conference on Multimedia. pp. 4492\u20134501.","DOI":"10.1145\/3581783.3611709"},{"issue":"6","key":"10.1016\/j.cviu.2026.104741_b48","doi-asserted-by":"crossref","first-page":"2872","DOI":"10.1109\/TPAMI.2021.3054775","article-title":"Deep learning for person re-identification: A survey and outlook","volume":"44","author":"Ye","year":"2021","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104741_b49","doi-asserted-by":"crossref","unstructured":"Zhang,\u00a0Y., Lu,\u00a0H., 2018. Deep cross-modal projection learning for image-text matching. In: Proceedings of the European Conference on Computer Vision. ECCV, pp. 686\u2013701.","DOI":"10.1007\/978-3-030-01246-5_42"},{"issue":"2","key":"10.1016\/j.cviu.2026.104741_b50","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3383184","article-title":"Dual-path convolutional image-text embeddings with instance loss","volume":"16","author":"Zheng","year":"2020","journal-title":"ACM Trans. Multimed. Comput. Commun. Appl. (TOMM)"},{"issue":"2","key":"10.1016\/j.cviu.2026.104741_b51","doi-asserted-by":"crossref","first-page":"5752","DOI":"10.1109\/TCE.2025.3570181","article-title":"AFES: Attention-based feature excitation and sorting for action recognition","volume":"71","author":"Zhou","year":"2025","journal-title":"IEEE Trans. Consum. Electron."},{"key":"10.1016\/j.cviu.2026.104741_b52","doi-asserted-by":"crossref","unstructured":"Zhu,\u00a0A., Wang,\u00a0Z., Li,\u00a0Y., Wan,\u00a0X., Jin,\u00a0J., Wang,\u00a0T., Hu,\u00a0F., Hua,\u00a0G., 2021. DSSL: Deep Surroundings-person Separation Learning for Text-based Person Retrieval. In: Proceedings of the 29th ACM International Conference on Multimedia. pp. 209\u2013217.","DOI":"10.1145\/3474085.3475369"},{"key":"10.1016\/j.cviu.2026.104741_b53","series-title":"Plip: Language-image pre-training for person representation learning","author":"Zuo","year":"2023"},{"key":"10.1016\/j.cviu.2026.104741_b54","doi-asserted-by":"crossref","unstructured":"Zuo,\u00a0J., Zhou,\u00a0H., Nie,\u00a0Y., Zhang,\u00a0F., Guo,\u00a0T., Sang,\u00a0N., Wang,\u00a0Y., Gao,\u00a0C., 2024. UFineBench: Towards Text-based Person Retrieval with Ultra-fine Granularity. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 22010\u201322019.","DOI":"10.1109\/CVPR52733.2024.02078"}],"container-title":["Computer Vision and Image Understanding"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1077314226001086?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1077314226001086?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,4,14]],"date-time":"2026-04-14T22:13:50Z","timestamp":1776204830000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1077314226001086"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4]]},"references-count":54,"alternative-id":["S1077314226001086"],"URL":"https:\/\/doi.org\/10.1016\/j.cviu.2026.104741","relation":{},"ISSN":["1077-3142"],"issn-type":[{"value":"1077-3142","type":"print"}],"subject":[],"published":{"date-parts":[[2026,4]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"CLIP-driven fine-grained mining for text-based person search","name":"articletitle","label":"Article Title"},{"value":"Computer Vision and Image Understanding","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.cviu.2026.104741","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Published by Elsevier Inc.","name":"copyright","label":"Copyright"}],"article-number":"104741"}}