{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T21:16:15Z","timestamp":1783199775473,"version":"3.54.6"},"reference-count":48,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100004028","name":"Jiangnan University","doi-asserted-by":"publisher","award":["22671037"],"award-info":[{"award-number":["22671037"]}],"id":[{"id":"10.13039\/501100004028","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62173160"],"award-info":[{"award-number":["62173160"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.eswa.2026.132693","type":"journal-article","created":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T15:03:14Z","timestamp":1777647794000},"page":"132693","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["BiMSD: Bidirectional masked similarity distillation with interpretability for visible-infrared person re-identification"],"prefix":"10.1016","volume":"325","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-9694-0641","authenticated-orcid":false,"given":"Yiqin","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1674-0869","authenticated-orcid":false,"given":"Ying","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.132693_bib0001","doi-asserted-by":"crossref","first-page":"2352","DOI":"10.1109\/TIP.2022.3141868","article-title":"Structure-aware positional transformer for visible-infrared person re-identification","volume":"31","author":"Chen","year":"2022","journal-title":"IEEE Transactions on Image Processing"},{"key":"10.1016\/j.eswa.2026.132693_bib0002","series-title":"International conference on machine learning","first-page":"1597","article-title":"A simple framework for contrastive learning of visual representations","author":"Chen","year":"2020"},{"key":"10.1016\/j.eswa.2026.132693_bib0003","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"15750","article-title":"Exploring simple siamese representation learning","author":"Chen","year":"2021"},{"key":"10.1016\/j.eswa.2026.132693_bib0004","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"587","article-title":"Neural feature search for RGB-infrared person re-identification","author":"Chen","year":"2021"},{"key":"10.1016\/j.eswa.2026.132693_bib0005","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"10257","article-title":"Hi-CMD: Hierarchical cross-modality disentanglement for visible-infrared person re-identification","author":"Choi","year":"2020"},{"key":"10.1016\/j.eswa.2026.132693_bib0006","article-title":"Bridging the safety-specific language model gap: domain-adaptive pretraining of transformer-based models across several industrial sectors for occupational safety applications","volume":"299","author":"Danish","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132693_bib0007","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2025.114525","article-title":"A dual-aligned knowledge self-distillation framework for visible-infrared cross-modal person re-identification","volume":"330","author":"Deng","year":"2025","journal-title":"Knowledge-Based Systems"},{"issue":"1","key":"10.1016\/j.eswa.2026.132693_bib0008","doi-asserted-by":"crossref","first-page":"279","DOI":"10.1007\/s00371-020-02015-z","article-title":"Modality-transfer generative adversarial network and dual-level unified latent representation for visible thermal person re-identification","volume":"38","author":"Fan","year":"2022","journal-title":"The Visual Computer"},{"key":"10.1016\/j.eswa.2026.132693_bib0009","series-title":"Proceedings of the 29th ACM international conference on multimedia","first-page":"5257","article-title":"Mso: Multi-feature space joint optimization network for rgb-infrared person re-identification","author":"Gao","year":"2021"},{"key":"10.1016\/j.eswa.2026.132693_bib0010","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2025.126645","article-title":"Cfet: A cross-fusion enhanced transformer for visible-infrared person re-identification","volume":"271","author":"Guo","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132693_bib0011","unstructured":"Hinton, G., Vinyals, O., & Dean, J. (2015). Distilling the knowledge in a neural network. arXiv preprint arXiv: 1503.02531."},{"key":"10.1016\/j.eswa.2026.132693_bib0012","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2024.102429","article-title":"Fusion-driven deep feature network for enhanced object detection and tracking in video surveillance systems","volume":"109","author":"Jain","year":"2024","journal-title":"Information Fusion"},{"key":"10.1016\/j.eswa.2026.132693_bib0013","doi-asserted-by":"crossref","first-page":"8432","DOI":"10.1109\/TMM.2023.3237155","article-title":"Cross-modality transformer with modality mining for visible-infrared person re-identification","volume":"25","author":"Liang","year":"2023","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.eswa.2026.132693_bib0014","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"20973","article-title":"Learning modal-invariant and temporal-memory for video-based visible-infrared person re-identification","author":"Lin","year":"2022"},{"key":"10.1016\/j.eswa.2026.132693_bib0015","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"1835","article-title":"Learning progressive modality-shared transformers for effective visible-infrared person re-identification","volume":"vol. 37","author":"Lu","year":"2023"},{"key":"10.1016\/j.eswa.2026.132693_bib0016","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2025.128311","article-title":"A comprehensive review of on-board action recognition models in public transportation systems","volume":"290","author":"Meurie","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132693_bib0017","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"12046","article-title":"Learning by aligning: Visible-infrared person re-identification using cross-modal correspondences","author":"Park","year":"2021"},{"key":"10.1016\/j.eswa.2026.132693_bib0018","doi-asserted-by":"crossref","DOI":"10.1109\/TIFS.2026.3671087","article-title":"Bi-level inter-modality modulation for unsupervised visible-infrared person re-identification","author":"Peng","year":"2026","journal-title":"IEEE Transactions on Information Forensics and Security"},{"issue":"7","key":"10.1016\/j.eswa.2026.132693_bib0019","first-page":"1","article-title":"ReFID: Reciprocal frequency-aware generalizable person re-identification via decomposition and filtering","volume":"20","author":"Peng","year":"2024","journal-title":"ACM Transactions on Multimedia Computing, Communications and Applications"},{"key":"10.1016\/j.eswa.2026.132693_bib0020","series-title":"International conference on machine learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.eswa.2026.132693_bib0021","article-title":"Two-stage knowledge distillation for visible-infrared person re-identification","author":"Shi","year":"2025","journal-title":"Pattern Recognition"},{"key":"10.1016\/j.eswa.2026.132693_bib0022","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2025.126817","article-title":"Analysis of convolutional-based variational autoencoders for privacy protection in realtime video surveillance","volume":"274","author":"Sivalakshmi","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132693_bib0023","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2025.103256","article-title":"Federated weakly-supervised video anomaly detection with mixture of local-to-global experts","author":"Su","year":"2025","journal-title":"Information Fusion"},{"key":"10.1016\/j.eswa.2026.132693_bib0024","series-title":"Proceedings of the 30th ACM international conference on multimedia","first-page":"5333","article-title":"Not all pixels are matched: Dense contrastive learning for cross-modality person re-identification","volume":"1","author":"Sun","year":"2022"},{"key":"10.1016\/j.eswa.2026.132693_bib0025","doi-asserted-by":"crossref","first-page":"2800","DOI":"10.1109\/TIFS.2024.3354377","article-title":"Robust visible-infrared person re-identification based on polymorphic mask and wavelet graph convolutional network","volume":"19","author":"Sun","year":"2024","journal-title":"IEEE Transactions on Information Forensics and Security"},{"key":"10.1016\/j.eswa.2026.132693_bib0026","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision workshops","article-title":"Distance based training for cross-modality person re-identification","author":"Tekeli","year":"2019"},{"key":"10.1016\/j.eswa.2026.132693_bib0027","article-title":"Tensorized parameter-free multi-view spectral clustering based on fair representation learning","author":"Wang","year":"2026","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.eswa.2026.132693_bib0028","doi-asserted-by":"crossref","first-page":"9385","DOI":"10.1109\/TMM.2025.3613125","article-title":"Tensor completion framework by graph refinement for incomplete multi-view clustering","volume":"27","author":"Wang","year":"2025","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.eswa.2026.132693_bib0029","doi-asserted-by":"crossref","DOI":"10.1016\/j.engappai.2025.110850","article-title":"Fully logits guided distillation with intermediate decision learning for deep model compression","volume":"153","author":"Wang","year":"2025","journal-title":"Engineering Applications of Artificial Intelligence"},{"key":"10.1016\/j.eswa.2026.132693_bib0030","doi-asserted-by":"crossref","DOI":"10.1016\/j.aei.2025.104051","article-title":"Temperature-driven category decoupled knowledge distillation with interpretability for model compression","volume":"69","author":"Wang","year":"2026","journal-title":"Advanced Engineering Informatics"},{"key":"10.1016\/j.eswa.2026.132693_bib0031","article-title":"End-to-end open-vocabulary video visual relationship detection using multi-modal prompting","author":"Wang","year":"2025","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.132693_bib0032","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"618","article-title":"Learning to reduce dual-level discrepancy for infrared-visible person re-identification","volume":"1","author":"Wang","year":"2019"},{"key":"10.1016\/j.eswa.2026.132693_bib0033","series-title":"Proceedings of the IEEE international conference on computer vision","first-page":"5380","article-title":"Rgb-infrared cross-modality person re-identification","author":"Wu","year":"2017"},{"key":"10.1016\/j.eswa.2026.132693_bib0034","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"7031","article-title":"Cora: Adapting clip for open-vocabulary detection with region prompting and anchor pre-matching","author":"Wu","year":"2023"},{"key":"10.1016\/j.eswa.2026.132693_bib0035","doi-asserted-by":"crossref","first-page":"35","DOI":"10.1016\/j.neucom.2021.02.088","article-title":"Visible-infrared person re-identification with data augmentation via cycle-consistent adversarial network","volume":"443","author":"Xia","year":"2021","journal-title":"Neurocomputing"},{"key":"10.1016\/j.eswa.2026.132693_bib0036","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"11069","article-title":"Towards grand unified representation learning for unsupervised visible-infrared person re-identification","author":"Yang","year":"2023"},{"key":"10.1016\/j.eswa.2026.132693_bib0037","doi-asserted-by":"crossref","first-page":"6273","DOI":"10.1109\/TMM.2023.3347855","article-title":"Ssrr: Structural semantic representation reconstruction for visible-infrared person re-identification","volume":"26","author":"Yang","year":"2023","journal-title":"IEEE Transactions on Multimedia"},{"issue":"4","key":"10.1016\/j.eswa.2026.132693_bib0038","doi-asserted-by":"crossref","first-page":"2299","DOI":"10.1109\/TPAMI.2023.3332875","article-title":"Channel augmentation for visible-infrared re-identification","volume":"46","author":"Ye","year":"2023","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.132693_bib0039","unstructured":"Yu, X., Dong, N., Zhu, L., Peng, H., & Tao, D. (2024). Clip-driven semantic discovery network for visible-infrared person re-identification. arXiv preprint arXiv: 2401.05806."},{"key":"10.1016\/j.eswa.2026.132693_bib0040","article-title":"Frozen CLIP-DINO: A strong backbone for weakly supervised semantic segmentation","author":"Zhang","year":"2025","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"issue":"8","key":"10.1016\/j.eswa.2026.132693_bib0041","doi-asserted-by":"crossref","first-page":"5361","DOI":"10.1109\/TCSVT.2022.3144775","article-title":"Dual mutual learning for cross-modality person re-identification","volume":"32","author":"Zhang","year":"2022","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.eswa.2026.132693_bib0042","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"7349","article-title":"Fmcnet: Feature-level modality compensation for visible-infrared person re-identification","author":"Zhang","year":"2022"},{"key":"10.1016\/j.eswa.2026.132693_bib0043","first-page":"1","article-title":"Modality confusion learning: A versatile framework for visible-infrared re-identification","author":"Zhang","year":"2025","journal-title":"International Journal of Computer Vision"},{"key":"10.1016\/j.eswa.2026.132693_bib0044","doi-asserted-by":"crossref","first-page":"155","DOI":"10.1016\/j.patrec.2021.07.006","article-title":"Rgb-ir cross-modality person reid based on teacher-student gan model","volume":"150","author":"Zhang","year":"2021","journal-title":"Pattern Recognition Letters"},{"issue":"3","key":"10.1016\/j.eswa.2026.132693_bib0045","doi-asserted-by":"crossref","first-page":"1418","DOI":"10.1109\/TCSVT.2021.3072171","article-title":"Grayscale enhancement colorization network for visible-infrared person re-identification","volume":"32","author":"Zhong","year":"2021","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.eswa.2026.132693_bib0046","series-title":"Proceedings of the 2020 international conference on multimedia retrieval","first-page":"421","article-title":"Visible-infrared person re-identification via colorization-based siamese generative adversarial network","author":"Zhong","year":"2020"},{"key":"10.1016\/j.eswa.2026.132693_bib0047","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"11175","article-title":"Zegclip: Towards adapting clip for zero-shot semantic segmentation","author":"Zhou","year":"2023"},{"key":"10.1016\/j.eswa.2026.132693_bib0048","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2025.102979","article-title":"Modality-perceptive harmonization network for visible-infrared person re-identification","volume":"118","author":"Zuo","year":"2025","journal-title":"Information Fusion"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426016064?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426016064?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T20:32:33Z","timestamp":1783197153000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426016064"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":48,"alternative-id":["S0957417426016064"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132693","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"BiMSD: Bidirectional masked similarity distillation with interpretability for visible-infrared person re-identification","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132693","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"132693"}}