{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T16:06:56Z","timestamp":1784131616436,"version":"3.55.0"},"reference-count":85,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2025,3,18]],"date-time":"2025-03-18T00:00:00Z","timestamp":1742256000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,3,18]],"date-time":"2025-03-18T00:00:00Z","timestamp":1742256000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2025,7]]},"DOI":"10.1007\/s11263-025-02396-5","type":"journal-article","created":{"date-parts":[[2025,3,18]],"date-time":"2025-03-18T15:10:45Z","timestamp":1742310645000},"page":"4690-4711","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Fusion for Visual-Infrared Person ReID in Real-World Surveillance Using Corrupted Multimodal Data"],"prefix":"10.1007","volume":"133","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4362-4898","authenticated-orcid":false,"given":"Arthur","family":"Josi","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6258-8362","authenticated-orcid":false,"given":"Mahdi","family":"Alehdaghi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9446-1040","authenticated-orcid":false,"given":"Rafael M. O.","family":"Cruz","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6116-7945","authenticated-orcid":false,"given":"Eric","family":"Granger","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,3,18]]},"reference":[{"key":"2396_CR1","doi-asserted-by":"crossref","unstructured":"Alehdaghi, M., Josi, A., Cruz, R., Granger, E. (2022). Visible-infrared person re-identification using privileged intermediate information. arxiv:2209.09348.","DOI":"10.1007\/978-3-031-25072-9_48"},{"key":"2396_CR2","volume-title":"Multimodal machine learning: A survey and taxonomy","author":"T Baltru\u0161aitis","year":"2018","unstructured":"Baltru\u0161aitis, T., Ahuja, C., & Morency, L.-P. (2018). Multimodal machine learning: A survey and taxonomy. TPAMI: IEEE."},{"key":"2396_CR3","doi-asserted-by":"crossref","unstructured":"Bhuiyan, A., Liu, Y., Siva, P., Javan, M., Ayed, I.B., Granger, E. (2020). Pose guided gated fusion for person re-identification. WACV.","DOI":"10.1109\/WACV45572.2020.9093370"},{"key":"2396_CR4","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-019-01276-z","volume-title":"Siamese dense network for reflection removal with flash and no-flash image pairs","author":"Y Chang","year":"2020","unstructured":"Chang, Y., Jung, C., Sun, J., & Wang, F. (2020). Siamese dense network for reflection removal with flash and no-flash image pairs. IJCV: Springer."},{"key":"2396_CR5","doi-asserted-by":"crossref","unstructured":"Chen, J., Yang, Q., Meng, J., Zheng, W.-S., Lai, J.-H. (2019). Contour-guided person re-identification. PRCV.","DOI":"10.1007\/978-3-030-31726-3_25"},{"key":"2396_CR6","doi-asserted-by":"crossref","unstructured":"Chen, L.-C., Zhu, Y., Papandreou, G., Schroff, F., Adam, H. (2018). Encoder-decoder with atrous separable convolution for semantic image segmentation. ECCV.","DOI":"10.1007\/978-3-030-01234-2_49"},{"key":"2396_CR7","unstructured":"Chen, M., Wang, Z., Zheng, F. (2021). Benchmarks for corruption invariant person re-identification. arxiv:2111.00880."},{"key":"2396_CR8","doi-asserted-by":"crossref","unstructured":"Choi, S., Lee, S., Kim, Y., Kim, T., & Kim, C. (2020). Hi-cmd: hierarchical cross-modality disentanglement for visible-infrared person re-identification. CVPR.","DOI":"10.1109\/CVPR42600.2020.01027"},{"key":"2396_CR9","doi-asserted-by":"crossref","unstructured":"Ciregan, D., Meier, U., & Schmidhuber, J. (2012). Multi-column deep neural networks for image classification. CVPR.","DOI":"10.1109\/CVPR.2012.6248110"},{"key":"2396_CR10","doi-asserted-by":"crossref","unstructured":"Fu, D., Chen, D., Bao, J., Yang, H., Yuan, L., Zhang, L., & Chen, D. (2021). Unsupervised pre-training for person re-identification. CVPR.","DOI":"10.1109\/CVPR46437.2021.01451"},{"key":"2396_CR11","doi-asserted-by":"crossref","unstructured":"Gabeur, V., Nagrani, A., Sun, C., Alahari, K., & Schmid, C. (2022). Masking modalities for cross-modal video retrieval. WACV.","DOI":"10.1109\/WACV51458.2022.00217"},{"key":"2396_CR12","unstructured":"Geirhos, R., Temme, C.R., Rauber, J., Sch\u00fctt, H.H., Bethge, M., & Wichmann, F.A. (2018). Generalisation in humans and deep neural networks. NIPS."},{"key":"2396_CR13","unstructured":"Gong, Y., Zeng, Z., Chen, L., Luo, Y., Weng, B., & Ye, F. (2021). A person re-identification data augmentation method with adversarial defense effect. arxiv:2101.08783."},{"key":"2396_CR14","unstructured":"Gu, Y., Yang, K., Fu, S., Chen, S., Li, X., & Marsic, I. (2018). Hybrid attention based multimodal network for spoken language classification. Proceedings of the conference. association for computational linguistics. meeting."},{"key":"2396_CR15","unstructured":"Han, K., Wang, Y., Chen, H., Chen, X., Guo, J., Liu, Z., others (2020). A survey on visual transformer. arxiv:2012.12556."},{"key":"2396_CR16","doi-asserted-by":"crossref","unstructured":"Hao, X., Zhu, Y., Appalaraju, S., Zhang, A., Zhang, W., Li, B., & Li, M. (2022). Mixgen: A new multi-modal data augmentation. arxiv:2206.08358.","DOI":"10.1109\/WACVW58289.2023.00042"},{"key":"2396_CR17","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016). Deep residual learning for image recognition. CVPR.","DOI":"10.1109\/CVPR.2016.90"},{"key":"2396_CR18","doi-asserted-by":"crossref","unstructured":"He, S., Luo, H., Wang, P., Wang, F., Li, H., & Jiang, W. (2021). Transreid: Transformer-based object re-identification. arxiv:2102.04378.","DOI":"10.1109\/ICCV48922.2021.01474"},{"key":"2396_CR19","doi-asserted-by":"crossref","unstructured":"Hendrycks, D., Basart, S., Mu, N., Kadavath, S., Wang, F., & Dorundo, E. (2021). others. A critical analysis of out-of-distribution generalization. ICCV: The many faces of robustness.","DOI":"10.1109\/ICCV48922.2021.00823"},{"key":"2396_CR20","unstructured":"Hendrycks, D., & Dietterich, T. (2019). Benchmarking neural network robustness to common corruptions and perturbations. arxiv:1903.12261."},{"key":"2396_CR21","doi-asserted-by":"crossref","unstructured":"Hendrycks, D., Liu, X., Wallace, E., Dziedzic, A., Krishnan, R., & Song, D. (2020). Pretrained transformers improve out-of-distribution robustness. arxiv:2004.06100.","DOI":"10.18653\/v1\/2020.acl-main.244"},{"key":"2396_CR22","unstructured":"Hendrycks, D., Mu, N., Cubuk, E.D., Zoph, B., Gilmer, J., & Lakshminarayanan, B. (2019). Augmix: A simple data processing method to improve robustness and uncertainty. arxiv:1912.02781."},{"key":"2396_CR23","unstructured":"Hermans, A., Beyer, L., & Leibe, B. (2017). In defense of the triplet loss for person re-identification. arxiv:1703.07737."},{"key":"2396_CR24","doi-asserted-by":"crossref","unstructured":"Hong, J., Kim, M., Choi, J., & Ro, Y.M. (2023). Watch or listen: Robust audio-visual speech recognition with visual corruption modeling and reliability scoring.","DOI":"10.1109\/CVPR52729.2023.01801"},{"key":"2396_CR25","unstructured":"Ismail, A.A., Hasan, M., & Ishtiaq, F. (2020). Improving multimodal accuracy through modality pre-training and attention. arxiv:2011.06102."},{"key":"2396_CR26","doi-asserted-by":"crossref","unstructured":"Josi, A., Alehdaghi, M., Cruz, R.M., & Granger, E. (2023). Multimodal data augmentation for visual-infrared person reid with corrupted data. WACV.","DOI":"10.1109\/WACVW58289.2023.00008"},{"key":"2396_CR27","unstructured":"Joze, H.R.V., Shaban, A., Iuzzolino, M.L., & Koishida, K. (2020). Mmtm: multimodal transfer module for cnn fusion. CVPR."},{"key":"2396_CR28","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2019.03.001","volume-title":"A survey of advances in vision-based vehicle re-identification","author":"SD Khan","year":"2019","unstructured":"Khan, S. D., & Ullah, H. (2019). A survey of advances in vision-based vehicle re-identification. CVIU: Elsevier."},{"key":"2396_CR29","doi-asserted-by":"crossref","unstructured":"Kniaz, V.V., Knyaz, V.A., Hladuvka, J., Kropatsch, W.G., & Mizginov, V. (2018). Thermalgan: Multimodal color-to-thermal image translation for person re-identification in multispectral dataset. ECCV workshops.","DOI":"10.1007\/978-3-030-11024-6_46"},{"key":"2396_CR30","doi-asserted-by":"publisher","first-page":"1497","DOI":"10.1109\/LSP.2023.3324552","volume":"30","author":"M Li","year":"2023","unstructured":"Li, M., Yang, D., & Zhang, L. (2023). Towards robust multimodal sentiment analysis under uncertain signal missing. IEEE Signal Processing Letters, 30, 1497\u20131501. https:\/\/doi.org\/10.1109\/LSP.2023.3324552","journal-title":"IEEE Signal Processing Letters"},{"key":"2396_CR31","doi-asserted-by":"crossref","unstructured":"Lian, Z., Liu, B., & Tao, J. (2021). Ctnet: Conversational transformer network for emotion recognition. TASLP: IEEE.","DOI":"10.1109\/TASLP.2021.3049898"},{"key":"2396_CR32","doi-asserted-by":"crossref","unstructured":"Lohweg, V., & M\u00f6nks, U. (2010). Fuzzy-pattern-classifier based sensor fusion for machine conditioning. Sensor fusion and its applications. C. Thomas, Ed. InTech.","DOI":"10.5772\/9969"},{"key":"2396_CR33","doi-asserted-by":"crossref","unstructured":"Luo, H., Gu, Y., Liao, X., Lai, S., & Jiang, W. (2019). Bag of tricks and a strong baseline for deep person re-identification. CVPR workshops.","DOI":"10.1109\/CVPRW.2019.00190"},{"key":"2396_CR34","doi-asserted-by":"crossref","unstructured":"Luo, H., Jiang, W., Gu, Y., Liu, F., Liao, X., Lai, S., & Gu, J. (2019). A strong baseline and batch normalization neck for deep person re-identification. IEEE transactions on multimedia: IEEE.","DOI":"10.1109\/TMM.2019.2958756"},{"key":"2396_CR35","doi-asserted-by":"crossref","unstructured":"Luo, W., Xing, J., Milan, A., Zhang, X., Liu, W., & Kim, T.-K. (2021). Multiple object tracking: A literature review. Artificial intelligence: Elsevier.","DOI":"10.1016\/j.artint.2020.103448"},{"key":"2396_CR36","doi-asserted-by":"crossref","unstructured":"Ma, M., Ren, J., Zhao, L., Tulyakov, S., Wu, C., & Peng, X. (2021). Smil: Multimodal learning with severely missing modality. Proceedings of the aaai conference on artificial intelligence (Vol.\u00a035, pp. 2302\u20132310).","DOI":"10.1609\/aaai.v35i3.16330"},{"key":"2396_CR37","doi-asserted-by":"crossref","unstructured":"Mekhazni, D., Bhuiyan, A., Ekladious, G., & Granger, E. (2020). Unsupervised domain adaptation in the dissimilarity space for person ReID. ECCV.","DOI":"10.1007\/978-3-030-58583-9_10"},{"key":"2396_CR38","unstructured":"Michaelis, C., Mitzkus, B., Geirhos, R., Rusak, E., Bringmann, O., Ecker, A.S., & Brendel, W. (2019). Benchmarking robustness in object detection: Autonomous driving when winter is coming. arxiv:1907.07484."},{"key":"2396_CR39","doi-asserted-by":"crossref","unstructured":"Mittal, A., Moorthy, A.K., & Bovik, A.C. (2011). Blind\/referenceless image spatial quality evaluator. ASILOMAR.","DOI":"10.1109\/ACSSC.2011.6190099"},{"key":"2396_CR40","unstructured":"M\u00fcller, R., Kornblith, S., & Hinton, G.E. (2019). When does label smoothing help? NeurIPS."},{"key":"2396_CR41","doi-asserted-by":"crossref","unstructured":"Nakamura, Y., Ishii, Y., Maruyama, Y., & Yamashita, T. (2022). Few-shot adaptive object detection with cross-domain cutmix. ACCV.","DOI":"10.1007\/978-3-031-26351-4_45"},{"key":"2396_CR42","doi-asserted-by":"crossref","unstructured":"Nguyen, D. T., Hong, H. G., Kim, K. W., & Park, K. R. (2017). Person recognition system based on a combination of body images from visible light and thermal cameras. Sensors: Multidisciplinary Digital Publishing Institute.","DOI":"10.3390\/s17030605"},{"key":"2396_CR43","unstructured":"Niu, S., Wu, J., Zhang, Y., Wen, Z., Chen, Y., Zhao, P., & Tan, M. (2023). Towards stable test-time adaptation in dynamic wild world. arXiv preprint arXiv:2302.12400"},{"key":"2396_CR44","doi-asserted-by":"crossref","unstructured":"Poria, S., Cambria, E., Bajpai, R., & Hussain, A. (2017). A review of affective computing: From unimodal analysis to multimodal fusion. (Vol.\u00a037, pp. 98\u2013125). Elsevier.","DOI":"10.1016\/j.inffus.2017.02.003"},{"key":"2396_CR45","volume-title":"Multimodal co-learning: challenges, applications with datasets, recent advances and future directions","author":"A Rahate","year":"2022","unstructured":"Rahate, A., Walambe, R., Ramanna, S., & Kotecha, K. (2022). Multimodal co-learning: challenges, applications with datasets, recent advances and future directions. Elsevier."},{"key":"2396_CR46","unstructured":"Raschka, S. (2018). Model evaluation, model selection, and algorithm selection in machine learning. arxiv:1811.12808."},{"key":"2396_CR47","doi-asserted-by":"crossref","unstructured":"Ristani, E., & Tomasi, C. (2018). Features for multi-target multi-camera tracking and re-identification. CVPR.","DOI":"10.1109\/CVPR.2018.00632"},{"key":"2396_CR48","doi-asserted-by":"crossref","unstructured":"Rusak, E., Schott, L., Zimmermann, R.S., Bitterwolf, J., Bringmann, O., Bethge, M., & Brendel, W. (2020). A simple way to make neural networks robust against diverse image corruptions. ECCV.","DOI":"10.1007\/978-3-030-58580-8_4"},{"key":"2396_CR49","doi-asserted-by":"crossref","unstructured":"Sen, P.C., Hajra, M., & Ghosh, M. (2020). Supervised classification algorithms in machine learning: A survey and review. Emerging technology in modelling and graphics: Proceedings of IEM graph 2018.","DOI":"10.1007\/978-981-13-7403-6_11"},{"key":"2396_CR50","unstructured":"Sharma, C., Kapil, S.R., & Chapman, D. (2021). Person re-identification with a locally aware transformer. arxiv:2106.03720."},{"key":"2396_CR51","doi-asserted-by":"crossref","unstructured":"Shin, I., Tsai, Y.-H., Zhuang, B., Schulter, S., Liu, B., Garg, S., & Yoon, K.-J. (2022). Mm-tta: multi-modal test-time adaptation for 3d semantic segmentation. Proceedings of the ieee\/cvf conference on computer vision and pattern recognition (pp. 16928\u201316937).","DOI":"10.1109\/CVPR52688.2022.01642"},{"key":"2396_CR52","doi-asserted-by":"crossref","unstructured":"Shorten, C., & Khoshgoftaar, T.M. (2019). A survey on image data augmentation for deep learning. Journal of big data. SpringerOpen.","DOI":"10.1186\/s40537-019-0197-0"},{"key":"2396_CR53","doi-asserted-by":"crossref","unstructured":"Snoek, C.G., Worring, M., & Smeulders, A.W. (2005). Early versus late fusion in semantic video analysis. ICMI.","DOI":"10.1145\/1101149.1101236"},{"key":"2396_CR54","doi-asserted-by":"crossref","unstructured":"Somers, V., De\u00a0Vleeschouwer, C., & Alahi, A. (2023). Body part-based representation learning for occluded person re-identification. WACV.","DOI":"10.1109\/WACV56688.2023.00166"},{"key":"2396_CR55","unstructured":"Su, L., Hu, C., Li, G., & Cao, D. (2020). Msaf: Multimodal split attention fusion. arxiv:2012.07175."},{"key":"2396_CR56","doi-asserted-by":"crossref","unstructured":"Sun, L., Liu, B., Tao, J., & Lian, Z. (2021). Multimodal cross-and self-attention network for speech emotion recognition. ICASSP.","DOI":"10.1109\/ICASSP39728.2021.9414654"},{"key":"2396_CR57","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Vanhoucke, V., Ioffe, S., Shlens, J., & Wojna, Z. (2016). Rethinking the inception architecture for computer vision. CVPR.","DOI":"10.1109\/CVPR.2016.308"},{"key":"2396_CR58","doi-asserted-by":"crossref","unstructured":"Wang, G., Zhang, T., Cheng, J., Liu, S., Yang, Y., & Hou, Z. (2019). Rgb-infrared cross-modality person re-identification via joint pixel and feature alignment. ICCV.","DOI":"10.1109\/ICCV.2019.00372"},{"key":"2396_CR59","doi-asserted-by":"crossref","unstructured":"Wang, H., Chen, Y., Ma, C., Avery, J., Hull, L., & Carneiro, G. (2023). Multi-modal learning with missing modality via shared-specific feature modelling. Proceedings of the ieee\/cvf conference on computer vision and pattern recognition (pp. 15878\u201315887).","DOI":"10.1109\/CVPR52729.2023.01524"},{"key":"2396_CR60","doi-asserted-by":"crossref","unstructured":"Wang, X., Shu, K., Kuang, H., Luo, S., Jin, R., & Liu, J. (2021). The role of spatial alignment in multimodal medical image fusion using deep learning for diagnostic problems. ICIMH.","DOI":"10.1145\/3484377.3484384"},{"key":"2396_CR61","doi-asserted-by":"crossref","unstructured":"Wang, Y. (2021). Survey on deep multi-modal data analytics: Collaboration, rivalry, and fusion. ACM New York, NY: TOMM.","DOI":"10.1145\/3408317"},{"key":"2396_CR62","doi-asserted-by":"crossref","unstructured":"Wang, Z., Li, C., Zheng, A., He, R., & Tang, J. (2022). Interact, embed, and enlarge: Boosting modality-specific representations for multi-modal person re-identification. Proceedings of the AAAI conference on artificial intelligence.","DOI":"10.1609\/aaai.v36i3.20165"},{"key":"2396_CR63","doi-asserted-by":"crossref","unstructured":"Wang, Z., Wang, Z., Zheng, Y., Chuang, Y.-Y., & Satoh, S. (2019). Learning to reduce dual-level discrepancy for infrared-visible person re-identification. CVPR.","DOI":"10.1109\/CVPR.2019.00071"},{"key":"2396_CR64","doi-asserted-by":"crossref","unstructured":"Wei, X., Zhang, T., Li, Y., Zhang, Y., & Wu, F. (2020). Multi-modality cross attention network for image and sentence matching. CVPR.","DOI":"10.1109\/CVPR42600.2020.01095"},{"key":"2396_CR65","doi-asserted-by":"crossref","unstructured":"Wu, A., Zheng, W.-S., Yu, H.-X., Gong, S., & Lai, J. (2017). Rgb-infrared cross-modality person re-identification. ICCV.","DOI":"10.1109\/ICCV.2017.575"},{"key":"2396_CR66","unstructured":"Xiao, T., Xia, T., Yang, Y., Huang, C., & Wang, X. (2015). Learning from massive noisy labeled data for image classification. Proceedings of the ieee conference on computer vision and pattern recognition (pp. 2691\u20132699)."},{"key":"2396_CR67","doi-asserted-by":"crossref","unstructured":"Xie, Q., Luong, M.-T., Hovy, E., & Le, Q.V. (2020). Self-training with noisy student improves imagenet classification. CVPR.","DOI":"10.1109\/CVPR42600.2020.01070"},{"issue":"6","key":"2396_CR68","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1109\/MIS.2020.3026715","volume":"36","author":"N Xu","year":"2020","unstructured":"Xu, N., Mao, W., Wei, P., & Zeng, D. (2020). Mda: Multimodal data augmentation framework for boosting performance on sentiment\/emotion classification tasks. IEEE intelligent systems., 36(6), 3\u201312.","journal-title":"IEEE intelligent systems."},{"issue":"9","key":"2396_CR69","doi-asserted-by":"publisher","first-page":"2499","DOI":"10.1109\/TMI.2022.3164050","volume":"41","author":"K Xuan","year":"2022","unstructured":"Xuan, K., Xiang, L., Huang, X., Zhang, L., Liao, S., Shen, D., & Wang, Q. (2022). Multimodal mri reconstruction assisted with spatial alignment network. IEEE Transactions on Medical Imaging., 41(9), 2499\u20132509.","journal-title":"IEEE Transactions on Medical Imaging."},{"key":"2396_CR70","doi-asserted-by":"crossref","unstructured":"Yang, M., Huang, Z., Hu, P., Li, T., Lv, J., Peng, X. (2022). Learning with twin noisy labels for visible-infrared person re-identification. Proceedings of the ieee\/cvf conference on computer vision and pattern recognition (pp. 14308\u201314317).","DOI":"10.1109\/CVPR52688.2022.01391"},{"issue":"7","key":"2396_CR71","doi-asserted-by":"publisher","first-page":"2511","DOI":"10.1007\/s11263-024-01997-w","volume":"132","author":"M Yang","year":"2024","unstructured":"Yang, M., Huang, Z., & Peng, X. (2024). Robust object re-identification with coupled noisy labels. International Journal of Computer Vision, 132(7), 2511\u20132529.","journal-title":"International Journal of Computer Vision"},{"issue":"6","key":"2396_CR72","first-page":"2872","volume":"44","author":"M Yang","year":"2024","unstructured":"Yang, M., Li, Y., Zhang, C., Hu, P., & Peng, X. (2024). Test-time adaptation against multi-modal reliability bias. The twelfth international conference on learning representations., 44(6), 2872\u20132893.","journal-title":"The twelfth international conference on learning representations."},{"key":"2396_CR73","volume-title":"Bi-directional center-constrained top-ranking for visible thermal person re-identification","author":"M Ye","year":"2019","unstructured":"Ye, M., Lan, X., Wang, Z., & Yuen, P. C. (2019). Bi-directional center-constrained top-ranking for visible thermal person re-identification. TIFS: IEEE."},{"key":"2396_CR74","doi-asserted-by":"publisher","first-page":"379","DOI":"10.1109\/TIP.2021.3131937","volume":"31","author":"M Ye","year":"2021","unstructured":"Ye, M., Li, H., Du, B., Shen, J., Shao, L., & Hoi, S. C. (2021). Collaborative refining for person re-identification with label noise. IEEE Transactions on Image Processing, 31, 379\u2013391.","journal-title":"IEEE Transactions on Image Processing"},{"key":"2396_CR75","unstructured":"Ye, M., Shen, J., Lin, G., Xiang, T., Shao, L., & Hoi, S.C. (2021). Deep learning for person re-identification: A survey and outlook. TPAMI."},{"key":"2396_CR76","doi-asserted-by":"publisher","first-page":"2655","DOI":"10.1109\/TIFS.2020.2970590","volume":"15","author":"M Ye","year":"2020","unstructured":"Ye, M., & Yuen, P. C. (2020). Purifynet: A robust person re-identification model with noisy labels. IEEE Transactions on Information Forensics and Security, 15, 2655\u20132666.","journal-title":"IEEE Transactions on Information Forensics and Security"},{"key":"2396_CR77","unstructured":"Yu, Y., Sheng, L., He, R., & Liang, J. (2023). Benchmarking test-time adaptation against distribution shifts in image classification. arXiv preprint arXiv:2307.03133"},{"key":"2396_CR78","doi-asserted-by":"publisher","DOI":"10.1016\/j.dsp.2022.103514","volume-title":"A survey of modern deep learning based object detection models","author":"SSA Zaidi","year":"2022","unstructured":"Zaidi, S. S. A., Ansari, M. S., Aslam, A., Kanwal, N., Asghar, M., & Lee, B. (2022). A survey of modern deep learning based object detection models. DSP: Elsevier."},{"key":"2396_CR79","unstructured":"Zhang, H., Wu, C., Zhang, Z., Zhu, Y., Lin, H., Zhang, Z. others (2020). Resnest: Split-attention networks. arxiv:2004.08955."},{"key":"2396_CR80","doi-asserted-by":"crossref","unstructured":"Zhang, Q., Lai, C., Liu, J., Huang, N., & Han, J. (2022). Fmcnet: Feature-level modality compensation for visible-infrared person re-identification. CVPR.","DOI":"10.1109\/CVPR52688.2022.00720"},{"key":"2396_CR81","doi-asserted-by":"crossref","unstructured":"Zhang, S., Zhang, S., Huang, T., Gao, W., & Tian, Q. (2017). Learning affective features with a hybrid deep model for audio-visual emotion recognition. TCSVT: IEEE.","DOI":"10.1109\/TCSVT.2017.2719043"},{"key":"2396_CR82","doi-asserted-by":"crossref","unstructured":"Zhang, Y., He, N., Yang, J., Li, Y., Wei, D., Huang, Y., & Zheng, Y. (2022). mmformer: Multimodal medical transformer for incomplete multimodal learning of brain tumor segmentation. International conference on medical image computing and computer-assisted intervention (pp. 107\u2013117).","DOI":"10.1007\/978-3-031-16443-9_11"},{"key":"2396_CR83","doi-asserted-by":"crossref","unstructured":"Zheng, A., Wang, Z., Chen, Z., Li, C., & Tang, J. (2021). Robust multi-modality person re-identification. Proceedings of the AAAI conference on artificial intelligence.","DOI":"10.1609\/aaai.v35i4.16467"},{"key":"2396_CR84","doi-asserted-by":"crossref","unstructured":"Zhong, Z., Zheng, L., Kang, G., Li, S., & Yang, Y. (2020). Random erasing data augmentation. Proceedings of the AAAI conference on artificial intelligence.","DOI":"10.1609\/aaai.v34i07.7000"},{"key":"2396_CR85","unstructured":"Zou, Z., Shi, Z., Guo, Y., & Ye, J. (2019). Object detection in 20 years: A survey. arxiv:1905.05055."}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02396-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-025-02396-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02396-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,7]],"date-time":"2025-06-07T06:02:24Z","timestamp":1749276144000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-025-02396-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,3,18]]},"references-count":85,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2025,7]]}},"alternative-id":["2396"],"URL":"https:\/\/doi.org\/10.1007\/s11263-025-02396-5","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,3,18]]},"assertion":[{"value":"4 April 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 February 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 March 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}