{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,2]],"date-time":"2026-04-02T12:14:21Z","timestamp":1775132061772,"version":"3.50.1"},"reference-count":46,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2026,2,3]],"date-time":"2026-02-03T00:00:00Z","timestamp":1770076800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,2,3]],"date-time":"2026-02-03T00:00:00Z","timestamp":1770076800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Science and Technology Plan Project of Liaoning Province","award":["2024-BSLH-017"],"award-info":[{"award-number":["2024-BSLH-017"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2026,4]]},"DOI":"10.1007\/s00530-025-02133-5","type":"journal-article","created":{"date-parts":[[2026,2,3]],"date-time":"2026-02-03T03:24:45Z","timestamp":1770089085000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Cross-modal local fine-grained feature localization and alignment for text-to-image person re-identification"],"prefix":"10.1007","volume":"32","author":[{"given":"Tiantian","family":"Yan","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiexiang","family":"Fang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuyang","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dongsheng","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,2,3]]},"reference":[{"key":"2133_CR1","doi-asserted-by":"crossref","unstructured":"Aggarwal, S., Radhakrishnan, V.B., Chakraborty, A.: Text-based person search via attribute-aided matching. In: Proceedings of the IEEE\/CVF winter conference on applications of computer vision, 2617\u20132625 (2020)","DOI":"10.1109\/WACV45572.2020.9093640"},{"key":"2133_CR2","doi-asserted-by":"crossref","unstructured":"Bai, Y., Cao, M., Gao, D., et\u00a0al.: Rasa: Relation and sensitivity aware representation learning for text-based person search. In: Proceedings of the Thirty-Second International Joint Conference on Artificial Intelligence, 555\u2013563 (2023)","DOI":"10.24963\/ijcai.2023\/62"},{"key":"2133_CR3","doi-asserted-by":"publisher","first-page":"4057","DOI":"10.1109\/TIP.2021.3068825","volume":"30","author":"Y Chen","year":"2021","unstructured":"Chen, Y., Huang, R., Chang, H., et al.: Cross-modal knowledge adaptation for language-based person search. IEEE Trans. Image Process. 30, 4057\u20134069 (2021)","journal-title":"IEEE Trans. Image Process."},{"key":"2133_CR4","doi-asserted-by":"publisher","first-page":"171","DOI":"10.1016\/j.neucom.2022.04.081","volume":"494","author":"Y Chen","year":"2022","unstructured":"Chen, Y., Zhang, G., Lu, Y., et al.: Tipcb: A simple but effective part-based convolutional baseline for text-based person search. Neurocomputing 494, 171\u2013181 (2022)","journal-title":"Neurocomputing"},{"key":"2133_CR5","unstructured":"Ding, Z., Ding, C., Shao, Z., et\u00a0al.: Semantically self-aligned network for text-to-image part-aware person re-identification. arXiv preprint arXiv:2107.12666 (2021)"},{"key":"2133_CR6","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., et\u00a0al.: An image is worth 16x16 words: Transformers for image recognition at scale. In: International Conference on Learning Representations (2021)"},{"key":"2133_CR7","doi-asserted-by":"crossref","unstructured":"Farooq, A., Awais, M., Kittler, J., et\u00a0al.: Axm-net: Implicit cross-modal feature alignment for person re-identification. In: Proceedings of the AAAI conference on artificial intelligence, 4477\u20134485 (2022)","DOI":"10.1609\/aaai.v36i4.20370"},{"key":"2133_CR8","doi-asserted-by":"publisher","first-page":"163","DOI":"10.1109\/TIP.2023.3337653","volume":"33","author":"S He","year":"2024","unstructured":"He, S., Luo, H., Jiang, W., et al.: Vgsg: Vision-guided semantic-group network for text-based person search. IEEE Trans. Image Process. 33, 163\u2013176 (2024)","journal-title":"IEEE Trans. Image Process."},{"key":"2133_CR9","doi-asserted-by":"publisher","first-page":"7699","DOI":"10.1109\/TMM.2022.3225754","volume":"25","author":"Z Ji","year":"2022","unstructured":"Ji, Z., Hu, J., Liu, D., et al.: Asymmetric cross-scale alignment for text-based person search. IEEE Trans. Multimedia 25, 7699\u20137709 (2022)","journal-title":"IEEE Trans. Multimedia"},{"key":"2133_CR10","doi-asserted-by":"crossref","unstructured":"Jiang, D., Ye, M.: Cross-modal implicit relation reasoning and aligning for text-to-image person retrieval. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2787\u20132797 (2023)","DOI":"10.1109\/CVPR52729.2023.00273"},{"key":"2133_CR11","doi-asserted-by":"crossref","unstructured":"Jing, Y., Si, C., Wang, J., et\u00a0al.: Pose-guided multi-granularity attention network for text-based person search. In: Proceedings of the AAAI Conference on Artificial Intelligence, 11189\u201311196 (2020)","DOI":"10.1609\/aaai.v34i07.6777"},{"key":"2133_CR12","unstructured":"Kenton, J.D.M.W.C., Toutanova, L.K.: Bert: Pre-training of deep bidirectional transformers for language understanding. In: Proceedings of NAACL-HLT, 2 (2019)"},{"key":"2133_CR13","unstructured":"Kingma, D.P., Ba, J.: Adam: A method for stochastic optimization. In: Bengio Y, LeCun Y (eds) 3rd International Conference on Learning Representations (2015)"},{"key":"2133_CR14","doi-asserted-by":"crossref","unstructured":"Lee, K.H., Chen, X., Hua, G., et\u00a0al.: Stacked cross attention for image-text matching. In: Proceedings of the European conference on computer vision (ECCV), 201\u2013216 (2018)","DOI":"10.1007\/978-3-030-01225-0_13"},{"key":"2133_CR15","unstructured":"Li, J., Li, D., Xiong, C., et\u00a0al.: Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation. In: International conference on machine learning, 12888\u201312900 (2022)"},{"key":"2133_CR16","doi-asserted-by":"crossref","unstructured":"Li, S., Xiao, T., Li, H., et\u00a0al.: Person search with natural language description. In: Proceedings of the IEEE conference on computer vision and pattern recognition, 1970\u20131979 (2017)","DOI":"10.1109\/CVPR.2017.551"},{"key":"2133_CR17","doi-asserted-by":"crossref","unstructured":"Li, S., Xu, X., Yang, Y., et\u00a0al.: Dcel: Deep cross-modal evidential learning for text-based person retrieval. In: Proceedings of the 31st ACM International Conference on Multimedia, 6292\u20136300 (2023)","DOI":"10.1145\/3581783.3612244"},{"key":"2133_CR18","doi-asserted-by":"publisher","first-page":"6609","DOI":"10.1109\/TMM.2024.3355644","volume":"26","author":"D Lin","year":"2024","unstructured":"Lin, D., Peng, Y., Meng, J., et al.: Cross-modal adaptive dual association for text-to-image person retrieval. IEEE Trans. Multimedia 26, 6609\u20136620 (2024)","journal-title":"IEEE Trans. Multimedia"},{"key":"2133_CR19","doi-asserted-by":"crossref","unstructured":"Liu, J., Zha, Z.J., Hong, R., et\u00a0al.: Deep adversarial graph attention convolution network for text-based person search. In: Proceedings of the 27th ACM international conference on multimedia, 665\u2013673 (2019)","DOI":"10.1145\/3343031.3350991"},{"issue":"3","key":"2133_CR20","doi-asserted-by":"publisher","first-page":"333","DOI":"10.1631\/FITEE.2300747","volume":"25","author":"Y Luo","year":"2024","unstructured":"Luo, Y., Yang, Y.: Large language model and domain-specific model collaboration for smart education. Frontiers of Information Technology & Electronic Engineering 25(3), 333\u2013341 (2024)","journal-title":"Frontiers of Information Technology & Electronic Engineering"},{"key":"2133_CR21","doi-asserted-by":"crossref","unstructured":"Luo, Y., Liu, P., Guan, T., et\u00a0al.: Significance-aware information bottleneck for domain adaptive semantic segmentation. In: Proceedings of the IEEE\/CVF international conference on computer vision, 6778\u20136787 (2019a)","DOI":"10.1109\/ICCV.2019.00688"},{"key":"2133_CR22","doi-asserted-by":"crossref","unstructured":"Luo, Y., Zheng, L., Guan, T., et\u00a0al.: Taking a closer look at domain shift: Category-level adversaries for semantics consistent domain adaptation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, 2507\u20132516 (2019b)","DOI":"10.1109\/CVPR.2019.00261"},{"issue":"8","key":"2133_CR23","first-page":"3940","volume":"44","author":"Y Luo","year":"2021","unstructured":"Luo, Y., Liu, P., Zheng, L., et al.: Category-level adversarial adaptation for semantic segmentation using purified features. IEEE Trans. Pattern Anal. Mach. Intell. 44(8), 3940\u20133956 (2021)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"1","key":"2133_CR24","doi-asserted-by":"publisher","first-page":"335","DOI":"10.1007\/s11263-024-02194-5","volume":"133","author":"Y Luo","year":"2025","unstructured":"Luo, Y., Liu, P., Yang, Y.: Kill two birds with one stone: Domain generalization for semantic segmentation via network pruning. Int. J. Comput. Vision 133(1), 335\u2013352 (2025)","journal-title":"Int. J. Comput. Vision"},{"key":"2133_CR25","doi-asserted-by":"crossref","unstructured":"Ma, Y., Sun, X., Ji, J., et\u00a0al.: Beat: Bi-directional one-to-many embedding alignment for text-based person retrieval. In: Proceedings of the 31st ACM International Conference on Multimedia, 4157\u20134168 (2023)","DOI":"10.1145\/3581783.3611768"},{"key":"2133_CR26","doi-asserted-by":"publisher","first-page":"5542","DOI":"10.1109\/TIP.2020.2984883","volume":"29","author":"K Niu","year":"2020","unstructured":"Niu, K., Huang, Y., Ouyang, W., et al.: Improving description-based person re-identification by multi-granularity image-text alignments. IEEE Trans. Image Process. 29, 5542\u20135556 (2020)","journal-title":"IEEE Trans. Image Process."},{"key":"2133_CR27","doi-asserted-by":"crossref","unstructured":"Qin, Y., Chen, Y.,Peng, D., et\u00a0al.: Noisy-correspondence learning for text-to-image person re-identification. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 27197\u201327206 (2024)","DOI":"10.1109\/CVPR52733.2024.02568"},{"key":"2133_CR28","doi-asserted-by":"crossref","unstructured":"Sarafianos, N., Xu, X., Kakadiaris, I.A.: Adversarial representation learning for text-to-image matching. In: Proceedings of the IEEE\/CVF international conference on computer vision, 5814\u20135824 (2019)","DOI":"10.1109\/ICCV.2019.00591"},{"key":"2133_CR29","doi-asserted-by":"crossref","unstructured":"Shao, Z., Zhang, X., Fang, M., et\u00a0al.: Learning granularity-unified representations for text-to-image person re-identification. In: Proceedings of the 30th acm international conference on multimedia, 5566\u20135574 (2022)","DOI":"10.1145\/3503161.3548028"},{"key":"2133_CR30","doi-asserted-by":"crossref","unstructured":"Shao, Z., Zhang, X., Ding, C., et\u00a0al.: Unified pre-training with pseudo texts for text-to-image person re-identification. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 11174\u201311184 (2023)","DOI":"10.1109\/ICCV51070.2023.01026"},{"key":"2133_CR31","doi-asserted-by":"crossref","unstructured":"Shen, F., Shu, X., Du, X., et\u00a0al.: Pedestrian-specific bipartite-aware similarity learning for text-based person retrieval. In: Proceedings of the 31st ACM International Conference on Multimedia, 8922\u20138931 (2023)","DOI":"10.1145\/3581783.3612009"},{"key":"2133_CR32","doi-asserted-by":"crossref","unstructured":"Suo, W., Sun, M., Niu, K., et\u00a0al.: A simple and robust correlation filtering method for text-based person search. In: European conference on computer vision, Springer, 726\u2013742 (2022)","DOI":"10.1007\/978-3-031-19833-5_42"},{"key":"2133_CR33","doi-asserted-by":"crossref","unstructured":"Tan, W., Ding, C., Jiang, J., et\u00a0al.: Harnessing the power of mllms for transferable text-to-image person reid. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 17127\u201317137 (2024)","DOI":"10.1109\/CVPR52733.2024.01621"},{"key":"2133_CR34","doi-asserted-by":"crossref","unstructured":"Wang, C., Luo, Z., Lin, Y., et\u00a0al.: Text-based person search via multi-granularity embedding learning. In: IJCAI, 1068\u20131074 (2021)","DOI":"10.24963\/ijcai.2021\/148"},{"key":"2133_CR35","doi-asserted-by":"crossref","unstructured":"Wang, Z., Fang, Z., Wang, J., et\u00a0al.: Vitaa: Visual-textual attributes alignment in person search by natural language. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XII 16, Springer, 402\u2013420 (2020)","DOI":"10.1007\/978-3-030-58610-2_24"},{"key":"2133_CR36","doi-asserted-by":"crossref","unstructured":"Wang, Z., Zhu, A., Xue, J., et\u00a0al.: Caibc: Capturing all-round information beyond color for text-based person retrieval. In: Proceedings of the 30th ACM international conference on multimedia, 5314\u20135322 (2022a)","DOI":"10.1145\/3503161.3548057"},{"key":"2133_CR37","doi-asserted-by":"crossref","unstructured":"Wang, Z., Zhu, A., Xue, J., et\u00a0al.: Look before you leap: Improving text-based person retrieval by learning a consistent cross-modal common manifold. In: Proceedings of the 30th ACM international conference on multimedia, 1984\u20131992 (2022b)","DOI":"10.1145\/3503161.3548166"},{"key":"2133_CR38","doi-asserted-by":"crossref","unstructured":"Wu, Y., Yan, Z., Han, X., et\u00a0al.: Lapscore: language-guided person search via color reasoning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 1624\u20131633 (2021)","DOI":"10.1109\/ICCV48922.2021.00165"},{"key":"2133_CR39","doi-asserted-by":"crossref","unstructured":"Yan, S., Dong, N., Liu, J., et\u00a0al.: Learning comprehensive representations with richer self for text-to-image person re-identification. In: Proceedings of the 31st ACM international conference on multimedia, 6202\u20136211 (2023a)","DOI":"10.1145\/3581783.3611832"},{"key":"2133_CR40","doi-asserted-by":"publisher","first-page":"6032","DOI":"10.1109\/TIP.2023.3327924","volume":"32","author":"S Yan","year":"2023","unstructured":"Yan, S., Dong, N., Zhang, L., et al.: Clip-driven fine-grained text-image person re-identification. IEEE Trans. Image Process. 32, 6032\u20136046 (2023)","journal-title":"IEEE Trans. Image Process."},{"key":"2133_CR41","doi-asserted-by":"crossref","unstructured":"Yang, S., Zhou, Y., Zheng, Z., et\u00a0al.: Towards unified text-based person retrieval: A large-scale multi-attribute and language search benchmark. In: Proceedings of the 31st ACM International Conference on Multimedia, 4492\u20134501 (2023)","DOI":"10.1145\/3581783.3611709"},{"key":"2133_CR42","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Lu, H.: Deep cross-modal projection learning for image-text matching. In: Proceedings of the European conference on computer vision (ECCV), 686\u2013701 (2018)","DOI":"10.1007\/978-3-030-01246-5_42"},{"key":"2133_CR43","doi-asserted-by":"crossref","unstructured":"Zheng, K., Liu, W., Liu, J., et\u00a0al.: Hierarchical gumbel attention network for text-based person search. In: Proceedings of the 28th ACM international conference on multimedia, 3441\u20133449 (2020a)","DOI":"10.1145\/3394171.3413864"},{"key":"2133_CR44","doi-asserted-by":"crossref","unstructured":"Zheng, Z., Zheng, L., Garrett, M., et\u00a0al.: Dual-path convolutional image-text embeddings with instance loss. ACM Transactions on Multimedia Computing, Communications, and Applications (TOMM) 16(2):1\u201323 (2020b)","DOI":"10.1145\/3383184"},{"key":"2133_CR45","doi-asserted-by":"crossref","unstructured":"Zhu, A., Wang, Z., Li, Y., et\u00a0al.: Dssl: Deep surroundings-person separation learning for text-based person retrieval. In: Proceedings of the 29th ACM international conference on multimedia, 209\u2013217 (2021)","DOI":"10.1145\/3474085.3475369"},{"key":"2133_CR46","doi-asserted-by":"crossref","unstructured":"Zuo, J., Zhou, H., Nie, Y., et\u00a0al.: Ufinebench: Towards text-based person retrieval with ultra-fine granularity. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 22010\u201322019 (2024)","DOI":"10.1109\/CVPR52733.2024.02078"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-025-02133-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-025-02133-5","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-025-02133-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,2]],"date-time":"2026-04-02T11:37:15Z","timestamp":1775129835000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-025-02133-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,2,3]]},"references-count":46,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2026,4]]}},"alternative-id":["2133"],"URL":"https:\/\/doi.org\/10.1007\/s00530-025-02133-5","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,2,3]]},"assertion":[{"value":"19 August 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"1 December 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 February 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"91"}}