{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,3]],"date-time":"2026-06-03T20:10:33Z","timestamp":1780517433518,"version":"3.54.1"},"reference-count":51,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100006469","name":"Fundo para o Desenvolvimento das Ci\u00eancias e da Tecnologia","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100006469","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2023YFC3321600"],"award-info":[{"award-number":["2023YFC3321600"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100021171","name":"Basic and Applied Basic Research Foundation of Guangdong Province","doi-asserted-by":"publisher","award":["2025A1515012281"],"award-info":[{"award-number":["2025A1515012281"]}],"id":[{"id":"10.13039\/501100021171","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100004733","name":"University of Macau","doi-asserted-by":"publisher","award":["MYRG-GRG2024-00077-FST-UMDF"],"award-info":[{"award-number":["MYRG-GRG2024-00077-FST-UMDF"]}],"id":[{"id":"10.13039\/501100004733","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,11]]},"DOI":"10.1016\/j.patcog.2026.113511","type":"journal-article","created":{"date-parts":[[2026,3,17]],"date-time":"2026-03-17T06:20:47Z","timestamp":1773728447000},"page":"113511","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":1,"special_numbering":"PA","title":["Minimizing the pretraining gap: Domain-aligned text-based person retrieval"],"prefix":"10.1016","volume":"179","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0409-1467","authenticated-orcid":false,"given":"Shuyu","family":"Yang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6596-8117","authenticated-orcid":false,"given":"Yaxiong","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-8960-8279","authenticated-orcid":false,"given":"Yongrui","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2136-3196","authenticated-orcid":false,"given":"Li","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2434-9050","authenticated-orcid":false,"given":"Zhedong","family":"Zheng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.patcog.2026.113511_bib0001","series-title":"CVPR","first-page":"1970","article-title":"Person search with natural language description","author":"Li","year":"2017"},{"key":"10.1016\/j.patcog.2026.113511_bib0002","series-title":"The Boundaries of Data","article-title":"Object re-identification: problems, algorithms and responsible research practice","author":"Zheng","year":"2024"},{"key":"10.1016\/j.patcog.2026.113511_bib0003","unstructured":"J. Lei, X. Chen, N. Zhang, M. Wang, M. Bansal, T.L. Berg, L. Yu, LoopITR: combining dual and cross encoder architectures for image-text retrieval, (2022). arXiv: 2203.05465."},{"key":"10.1016\/j.patcog.2026.113511_bib0004","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110481","article-title":"Text-based person search via cross-modal alignment learning","volume":"152","author":"Ke","year":"2024","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113511_bib0005","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2023.109636","article-title":"BDNet: a BERT-based dual-path network for text-to-image cross-modal person re-identification","volume":"141","author":"Liu","year":"2023","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113511_bib0006","series-title":"ECCV","first-page":"624","article-title":"See finer, see more: Implicit modality alignment for text-based person retrieval","author":"Shu","year":"2022"},{"key":"10.1016\/j.patcog.2026.113511_bib0007","series-title":"CVPR","first-page":"2787","article-title":"Cross-modal implicit relation reasoning and aligning for text-to-image person retrieval","author":"Jiang","year":"2023"},{"key":"10.1016\/j.patcog.2026.113511_bib0008","article-title":"RaSa: relation and sensitivity aware representation learning for text-based person search","author":"Bai","year":"2023","journal-title":"IJCAI"},{"key":"10.1016\/j.patcog.2026.113511_bib0009","series-title":"ACMMM","first-page":"4492","article-title":"Towards unified text-based person retrieval: a large-scale multi-attribute and language search benchmark","author":"Yang","year":"2023"},{"key":"10.1016\/j.patcog.2026.113511_bib0010","series-title":"ICCV","first-page":"3754","article-title":"Unlabeled samples generated by GAN improve the person re-identification baseline in vitro","author":"Zheng","year":"2017"},{"issue":"1","key":"10.1016\/j.patcog.2026.113511_bib0011","doi-asserted-by":"crossref","first-page":"53","DOI":"10.1109\/MSP.2017.2765202","article-title":"Generative adversarial networks: an overview","volume":"35","author":"Creswell","year":"2018","journal-title":"IEEE Signal Process. Mag."},{"key":"10.1016\/j.patcog.2026.113511_bib0012","series-title":"CVPR","first-page":"10684","article-title":"High-resolution image synthesis with latent diffusion models","author":"Rombach","year":"2022"},{"key":"10.1016\/j.patcog.2026.113511_bib0013","unstructured":"Stability, Stable Diffusion v1.5 Model Card, 2022. https:\/\/huggingface.co\/runwayml\/stable-diffusion-v1-5."},{"issue":"6","key":"10.1016\/j.patcog.2026.113511_bib0014","doi-asserted-by":"crossref","first-page":"7534","DOI":"10.1109\/TNNLS.2022.3214834","article-title":"Parameter-efficient person re-identification in the 3D space","volume":"35","author":"Zheng","year":"2022","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"10.1016\/j.patcog.2026.113511_bib0015","series-title":"ECCV","first-page":"38","article-title":"Grounding DINO: marrying DINO with grounded pre-training for open-set object detection","author":"Liu","year":"2025"},{"key":"10.1016\/j.patcog.2026.113511_bib0016","series-title":"ECCV","first-page":"686","article-title":"Deep cross-modal projection learning for image-text matching","author":"Zhang","year":"2018"},{"issue":"2","key":"10.1016\/j.patcog.2026.113511_bib0017","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3383184","article-title":"Dual-path convolutional image-text embeddings with instance loss","volume":"16","author":"Zheng","year":"2020","journal-title":"ACM Trans. Multimed. Comput. Commun. Appl."},{"key":"10.1016\/j.patcog.2026.113511_bib0018","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.111247","article-title":"Local-enhanced representation for text-based person search","volume":"161","author":"Zhang","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113511_bib0019","doi-asserted-by":"crossref","first-page":"151","DOI":"10.1016\/j.patcog.2019.06.006","article-title":"Improving person re-identification by attribute and identity learning","volume":"95","author":"Lin","year":"2019","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113511_bib0020","series-title":"ECCV","first-page":"402","article-title":"ViTAA: visual-textual attributes alignment in person search by natural language","author":"Wang","year":"2020"},{"key":"10.1016\/j.patcog.2026.113511_bib0021","first-page":"1","article-title":"Relation-aware aggregation network with auxiliary guidance for text-based person search","volume":"25","author":"Zeng","year":"2021","journal-title":"World Wide Web"},{"key":"10.1016\/j.patcog.2026.113511_bib0022","series-title":"ACMMM","first-page":"1984","article-title":"Look before you leap: improving text-based person retrieval by learning a consistent cross-modal common manifold","author":"Wang","year":"2022"},{"key":"10.1016\/j.patcog.2026.113511_bib0023","series-title":"ACMMM","first-page":"5314","article-title":"CAIBC: capturing all-round information beyond color for text-based person retrieval","author":"Wang","year":"2022"},{"key":"10.1016\/j.patcog.2026.113511_bib0024","series-title":"ICML","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.patcog.2026.113511_bib0025","doi-asserted-by":"crossref","first-page":"6032","DOI":"10.1109\/TIP.2023.3327924","article-title":"Clip-driven fine-grained text-image person re-identification","volume":"32","author":"Yan","year":"2023","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.patcog.2026.113511_bib0026","first-page":"9694","article-title":"Align before fuse: vision and language representation learning with momentum distillation","volume":"34","author":"Li","year":"2021","journal-title":"NeurIPS"},{"key":"10.1016\/j.patcog.2026.113511_bib0027","unstructured":"A. Hertz, R. Mokady, J. Tenenbaum, K. Aberman, Y. Pritch, D. Cohen-Or, Prompt-to-prompt image editing with cross attention control, (2022). arXiv: 2208.01626."},{"key":"10.1016\/j.patcog.2026.113511_bib0028","series-title":"CVPR","first-page":"2426","article-title":"DiffusionCLIP: text-guided diffusion models for robust image manipulation","author":"Kim","year":"2022"},{"key":"10.1016\/j.patcog.2026.113511_bib0029","series-title":"SIGGRAPH","first-page":"1","article-title":"Zero-shot image-to-image translation","author":"Parmar","year":"2023"},{"key":"10.1016\/j.patcog.2026.113511_bib0030","series-title":"CVPR","first-page":"22500","article-title":"DreamBooth: fine tuning text-to-image diffusion models for subject-driven generation","author":"Ruiz","year":"2023"},{"key":"10.1016\/j.patcog.2026.113511_bib0031","series-title":"ICCV","first-page":"3836","article-title":"Adding conditional control to text-to-image diffusion models","author":"Zhang","year":"2023"},{"key":"10.1016\/j.patcog.2026.113511_bib0032","doi-asserted-by":"crossref","unstructured":"S. Yang, Y. Wang, L. Zhu, Z. Zheng, Beyond walking: a large-scale image-text benchmark for text-based person anomaly search, (2024). arXiv: 2411.17776.","DOI":"10.1109\/ICCV51701.2025.01090"},{"key":"10.1016\/j.patcog.2026.113511_bib0033","unstructured":"S. Azizi, S. Kornblith, C. Saharia, M. Norouzi, D.J. Fleet, Synthetic data from diffusion models improves ImageNet classification, (2023). arXiv: 2304.08466."},{"key":"10.1016\/j.patcog.2026.113511_bib0034","series-title":"CVPR","first-page":"248","article-title":"ImageNet: a large-scale hierarchical image database","author":"Deng","year":"2009"},{"key":"10.1016\/j.patcog.2026.113511_bib0035","series-title":"MICCAI","first-page":"234","article-title":"U-net: convolutional networks for biomedical image segmentation","author":"Ronneberger","year":"2015"},{"key":"10.1016\/j.patcog.2026.113511_bib0036","series-title":"ICCV","first-page":"2223","article-title":"Unpaired image-to-image translation using cycle-consistent adversarial networks","author":"Zhu","year":"2017"},{"key":"10.1016\/j.patcog.2026.113511_bib0037","unstructured":"J. Wei, K. Zou, Eda: Easy data augmentation techniques for boosting performance on text classification tasks, (2019). arXiv: 1901.11196."},{"key":"10.1016\/j.patcog.2026.113511_bib0038","series-title":"CVPR","first-page":"7291","article-title":"Realtime multi-person 2D pose estimation using part affinity fields","author":"Cao","year":"2017"},{"key":"10.1016\/j.patcog.2026.113511_bib0039","series-title":"ICML","first-page":"19730","article-title":"BLIP-2: bootstrapping language-image pre-training with frozen image encoders and large language models","author":"Li","year":"2023"},{"key":"10.1016\/j.patcog.2026.113511_bib0040","series-title":"ICCV","first-page":"10012","article-title":"Swin transformer: hierarchical vision transformer using shifted windows","author":"Liu","year":"2021"},{"key":"10.1016\/j.patcog.2026.113511_bib0041","unstructured":"Z. Ding, C. Ding, Z. Shao, D. Tao, Semantically self-aligned network for text-to-image part-aware person re-identification, (2021). arXiv: 2107.12666."},{"key":"10.1016\/j.patcog.2026.113511_bib0042","series-title":"ACMMM","first-page":"209","article-title":"DSSL: deep surroundings-person separation learning for text-based person retrieval","author":"Zhu","year":"2021"},{"key":"10.1016\/j.patcog.2026.113511_bib0043","series-title":"CVPR","first-page":"79","article-title":"Person transfer GAN to bridge domain gap for person re-identification","author":"Wei","year":"2018"},{"key":"10.1016\/j.patcog.2026.113511_bib0044","doi-asserted-by":"crossref","first-page":"5542","DOI":"10.1109\/TIP.2020.2984883","article-title":"Improving description-based person re-identification by multi-granularity image-text alignments","volume":"29","author":"Niu","year":"2020","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.patcog.2026.113511_bib0045","series-title":"ACMMM","first-page":"5566","article-title":"Learning granularity-unified representations for text-to-image person re-identification","author":"Shao","year":"2022"},{"key":"10.1016\/j.patcog.2026.113511_bib0046","series-title":"ECCV","first-page":"201","article-title":"Stacked cross attention for image-text matching","author":"Lee","year":"2018"},{"key":"10.1016\/j.patcog.2026.113511_bib0047","series-title":"ICLR","article-title":"Decoupled weight decay regularization","author":"Loshchilov","year":"2019"},{"key":"10.1016\/j.patcog.2026.113511_bib0048","series-title":"NAACL","first-page":"2","article-title":"BERT: pre-training of deep bidirectional transformers for language understanding","volume":"Vol. 1","author":"Devlin","year":"2019"},{"key":"10.1016\/j.patcog.2026.113511_bib0049","series-title":"ICCV","first-page":"1116","article-title":"Scalable person re-identification: a benchmark","author":"Zheng","year":"2015"},{"key":"10.1016\/j.patcog.2026.113511_bib0050","series-title":"ECCV","first-page":"740","article-title":"Microsoft COCO: common objects in context","author":"Lin","year":"2014"},{"key":"10.1016\/j.patcog.2026.113511_bib0051","series-title":"ICCV","first-page":"11720","article-title":"Beyond walking: A large-scale image-text benchmark for text-based person anomaly search","author":"Yang","year":"2025"}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326004772?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326004772?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,3]],"date-time":"2026-06-03T13:10:23Z","timestamp":1780492223000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0031320326004772"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,11]]},"references-count":51,"alternative-id":["S0031320326004772"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113511","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,11]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Minimizing the pretraining gap: Domain-aligned text-based person retrieval","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113511","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"113511"}}