{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T20:18:03Z","timestamp":1783196283957,"version":"3.54.6"},"reference-count":55,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100021171","name":"Basic and Applied Basic Research Foundation of Guangdong Province","doi-asserted-by":"publisher","award":["2026A1515011198"],"award-info":[{"award-number":["2026A1515011198"]}],"id":[{"id":"10.13039\/501100021171","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62502317"],"award-info":[{"award-number":["62502317"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,12]]},"DOI":"10.1016\/j.patcog.2026.114038","type":"journal-article","created":{"date-parts":[[2026,5,23]],"date-time":"2026-05-23T15:07:19Z","timestamp":1779548839000},"page":"114038","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PB","title":["VSE-MOT: Multi-object tracking in low-quality video scenes guided by visual semantic enhancement"],"prefix":"10.1016","volume":"180","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-0383-0631","authenticated-orcid":false,"given":"Jun","family":"Du","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6378-926X","authenticated-orcid":false,"given":"Weiwei","family":"Xing","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ming","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fei Richard","family":"Yu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.patcog.2026.114038_b1","article-title":"Cbph-net: A small object detector for behavior recognition in classroom scenarios","author":"Zhao","year":"2023","journal-title":"IEEE Trans. Instrum. Meas."},{"key":"10.1016\/j.patcog.2026.114038_b2","doi-asserted-by":"crossref","DOI":"10.1109\/TIM.2023.3291011","article-title":"VIMOT: A tightly-coupled estimator for stereo visual-inertial navigation and multi-object tracking","author":"Feng","year":"2023","journal-title":"IEEE Trans. Instrum. Meas."},{"key":"10.1016\/j.patcog.2026.114038_b3","first-page":"1","article-title":"Lane-level and full-cycle multivehicle tracking using low-channel roadside LiDAR","volume":"72","author":"Liu","year":"2023","journal-title":"IEEE Trans. Instrum. Meas."},{"key":"10.1016\/j.patcog.2026.114038_b4","doi-asserted-by":"crossref","unstructured":"X. Wang, K. Ma, Q. Liu, Y. Zou, Y. Fu, Multi-Object Tracking in the Dark, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 382\u2013392.","DOI":"10.1109\/CVPR52733.2024.00044"},{"key":"10.1016\/j.patcog.2026.114038_b5","doi-asserted-by":"crossref","unstructured":"Y. Zhang, T. Wang, X. Zhang, Motrv2: Bootstrapping end-to-end multi-object tracking by pretrained object detectors, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 22056\u201322065.","DOI":"10.1109\/CVPR52729.2023.02112"},{"key":"10.1016\/j.patcog.2026.114038_b6","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110588","article-title":"Prototype learning based generic multiple object tracking via point-to-box supervision","volume":"154","author":"Liu","year":"2024","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114038_b7","series-title":"Crtrack: Low-light semi-supervised multi-object tracking based on consistency regularization","author":"Zhao","year":"2025"},{"key":"10.1016\/j.patcog.2026.114038_b8","doi-asserted-by":"crossref","unstructured":"C. Xiao, Q. Cao, Z. Luo, L. Lan, Mambatrack: a simple baseline for multiple object tracking with state space model, in: Proceedings of the 32nd ACM International Conference on Multimedia, 2024, pp. 4082\u20134091.","DOI":"10.1145\/3664647.3680944"},{"key":"10.1016\/j.patcog.2026.114038_b9","doi-asserted-by":"crossref","unstructured":"X. Wang, K. Ma, Q. Liu, Y. Zou, Y. Fu, Multi-object tracking in the dark, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 382\u2013392.","DOI":"10.1109\/CVPR52733.2024.00044"},{"key":"10.1016\/j.patcog.2026.114038_b10","series-title":"2016 IEEE International Conference on Image Processing","first-page":"3464","article-title":"Simple online and realtime tracking","author":"Bewley","year":"2016"},{"key":"10.1016\/j.patcog.2026.114038_b11","first-page":"41","article-title":"An introduction to the kalman filter","volume":"8","author":"Bishop","year":"2001"},{"key":"10.1016\/j.patcog.2026.114038_b12","series-title":"2017 IEEE International Conference on Image Processing","first-page":"3645","article-title":"Simple online and realtime tracking with a deep association metric","author":"Wojke","year":"2017"},{"key":"10.1016\/j.patcog.2026.114038_b13","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2022.109141","article-title":"Real-time siamese multiple object tracker with enhanced proposals","volume":"135","author":"Vaquero","year":"2023","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114038_b14","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.111091","article-title":"Seatrack: rethinking observation-centric sort for robust nearshore multiple object tracking","volume":"159","author":"Ding","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114038_b15","series-title":"European Conference on Computer Vision","first-page":"1","article-title":"Bytetrack: Multi-object tracking by associating every detection box","author":"Zhang","year":"2022"},{"key":"10.1016\/j.patcog.2026.114038_b16","series-title":"YOLOX: Exceeding YOLO series in 2021","author":"Zheng","year":"2021"},{"key":"10.1016\/j.patcog.2026.114038_b17","doi-asserted-by":"crossref","unstructured":"Y. Wang, Z. Xu, X. Wang, C. Shen, B. Cheng, H. Shen, H. Xia, End-to-end video instance segmentation with transformers, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2021, pp. 8741\u20138750.","DOI":"10.1109\/CVPR46437.2021.00863"},{"key":"10.1016\/j.patcog.2026.114038_b18","series-title":"European Conference on Computer Vision","first-page":"213","article-title":"End-to-end object detection with transformers","author":"Carion","year":"2020"},{"key":"10.1016\/j.patcog.2026.114038_b19","series-title":"European Conference on Computer Vision","first-page":"659","article-title":"Motr: End-to-end multiple-object tracking with transformer","author":"Zeng","year":"2022"},{"key":"10.1016\/j.patcog.2026.114038_b20","series-title":"Deformable detr: Deformable transformers for end-to-end object detection","author":"Zhu","year":"2020"},{"key":"10.1016\/j.patcog.2026.114038_b21","doi-asserted-by":"crossref","unstructured":"J. Cai, M. Xu, W. Li, Y. Xiong, W. Xia, Z. Tu, S. Soatto, Memot: Multi-object tracking with memory, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 8090\u20138100.","DOI":"10.1109\/CVPR52688.2022.00792"},{"key":"10.1016\/j.patcog.2026.114038_b22","series-title":"Transtrack: Multiple object tracking with transformer","author":"Sun","year":"2020"},{"key":"10.1016\/j.patcog.2026.114038_b23","series-title":"Multiple object tracking as ID prediction","author":"Gao","year":"2024"},{"key":"10.1016\/j.patcog.2026.114038_b24","article-title":"Multi-objective unlearning in recommender systems via preference guided pareto exploration","author":"Li","year":"2025","journal-title":"IEEE Transactions on Services Computing"},{"key":"10.1016\/j.patcog.2026.114038_b25","doi-asserted-by":"crossref","first-page":"12611","DOI":"10.52202\/075280-0553","article-title":"Ultrare: enhancing receraser for recommendation unlearning via error decomposition","volume":"36","author":"Li","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"2","key":"10.1016\/j.patcog.2026.114038_b26","doi-asserted-by":"crossref","first-page":"781","DOI":"10.1109\/TKDE.2025.3638174","article-title":"A survey on recommendation unlearning: fundamentals, taxonomy, evaluation, and open questions","volume":"38","author":"Li","year":"2025","journal-title":"IEEE Transactions on Knowledge and Data Engineering"},{"key":"10.1016\/j.patcog.2026.114038_b27","doi-asserted-by":"crossref","unstructured":"W. Yin, Y. Wang, G. Duan, D. Zhang, X. Hu, Y.-F. Li, T. He, Knowledge-Aligned Counterfactual-Enhancement Diffusion Perception for Unsupervised Cross-Domain Visual Emotion Recognition, in: IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), 2025, pp. 3888\u20133898.","DOI":"10.1109\/CVPR52734.2025.00368"},{"key":"10.1016\/j.patcog.2026.114038_b28","doi-asserted-by":"crossref","unstructured":"R. Dai, C. Li, Y. Yan, L. Mo, K. Qin, T. He, Unbiased Missing-modality Multimodal Learning, in: IEEE\/CVF International Conference on Computer Vision (ICCV), 2025, pp. 24507\u201324517.","DOI":"10.1109\/ICCV51701.2025.02272"},{"key":"10.1016\/j.patcog.2026.114038_b29","doi-asserted-by":"crossref","unstructured":"W. Yin, S. Zhan, C. Liu, X. Hu, G. Duan, X. Xie, Y.-F. Li, T. He, TiCAL: Typicality-Based Consistency-Aware Learning for Multimodal Emotion Recognition, in: AAAI Conference on Artificial Intelligence (AAAI), 2026.","DOI":"10.1609\/aaai.v40i21.38854"},{"key":"10.1016\/j.patcog.2026.114038_b30","article-title":"Unif2ace: fine-grained face understanding and generation with unified multimodal models","volume":"3","author":"Li","year":"2025","journal-title":"arXiv preprint arXiv:2503.08120"},{"key":"10.1016\/j.patcog.2026.114038_b31","series-title":"Proceedings of the 42nd International Conference on Machine Learning","first-page":"58621","article-title":"WMarkGPT: watermarked image understanding via multimodal large language models","volume":"267","author":"Tan","year":"2025"},{"key":"10.1016\/j.patcog.2026.114038_b32","doi-asserted-by":"crossref","unstructured":"Y. Shi, W. Yan, G. Xu, Y. Li, Y. Chen, Z. Li, F. Yu, M. Li, S.Y. Yeo, PVChat: Personalized Video Chat with One-Shot Learning, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), 2025, pp. 23321\u201323331.","DOI":"10.1109\/ICCV51701.2025.02165"},{"key":"10.1016\/j.patcog.2026.114038_b33","series-title":"VLA-ATTC: Adaptive Test-Time Compute for VLA Models with Relative Action Critic Model","author":"Li","year":"2026"},{"key":"10.1016\/j.patcog.2026.114038_b34","doi-asserted-by":"crossref","first-page":"293","DOI":"10.1016\/j.neucom.2022.07.028","article-title":"Clip4clip: An empirical study of clip for end to end video clip retrieval and captioning","volume":"508","author":"Luo","year":"2022","journal-title":"Neurocomputing"},{"key":"10.1016\/j.patcog.2026.114038_b35","series-title":"Actionclip: A new paradigm for video action recognition","author":"Wang","year":"2021"},{"key":"10.1016\/j.patcog.2026.114038_b36","series-title":"European Conference on Computer Vision","first-page":"681","article-title":"Zero-shot temporal action detection via vision-language prompting","author":"Nag","year":"2022"},{"key":"10.1016\/j.patcog.2026.114038_b37","first-page":"1405","article-title":"Clip-reid: exploiting vision-language model for image re-identification without concrete text labels","volume":"vol. 37","author":"Li","year":"2023"},{"key":"10.1016\/j.patcog.2026.114038_b38","doi-asserted-by":"crossref","DOI":"10.1016\/j.imavis.2025.105415","article-title":"Image re-identification: Where self-supervision meets vision-language learning","volume":"154","author":"Wang","year":"2025","journal-title":"Image Vis. Comput."},{"key":"10.1016\/j.patcog.2026.114038_b39","series-title":"2025 IEEE International Conference on Robotics and Automation","first-page":"6816","article-title":"Lamot: Language-guided multi-object tracking","author":"Li","year":"2025"},{"key":"10.1016\/j.patcog.2026.114038_b40","doi-asserted-by":"crossref","unstructured":"X. Wang, K. Ma, Q. Liu, Y. Zou, Y. Fu, Multi-object tracking in the dark, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 382\u2013392.","DOI":"10.1109\/CVPR52733.2024.00044"},{"key":"10.1016\/j.patcog.2026.114038_b41","doi-asserted-by":"crossref","unstructured":"S. Woo, J. Park, J.-Y. Lee, I.S. Kweon, Cbam: Convolutional block attention module, in: Proceedings of the European Conference on Computer Vision, ECCV, 2018, pp. 3\u201319.","DOI":"10.1007\/978-3-030-01234-2_1"},{"key":"10.1016\/j.patcog.2026.114038_b42","article-title":"Attention is all you need","author":"Vaswani","year":"2017","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"4","key":"10.1016\/j.patcog.2026.114038_b43","doi-asserted-by":"crossref","first-page":"834","DOI":"10.1109\/TPAMI.2017.2699184","article-title":"Deeplab: Semantic image segmentation with deep convolutional nets, atrous convolution, and fully connected crfs","volume":"40","author":"Chen","year":"2017","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.patcog.2026.114038_b44","doi-asserted-by":"crossref","unstructured":"P. Sun, J. Cao, Y. Jiang, Z. Yuan, S. Bai, K. Kitani, P. Luo, Dancetrack: Multi-object tracking in uniform appearance and diverse motion, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 20993\u201321002.","DOI":"10.1109\/CVPR52688.2022.02032"},{"key":"10.1016\/j.patcog.2026.114038_b45","series-title":"MOT16: A benchmark for multi-object tracking","author":"Milan","year":"2016"},{"key":"10.1016\/j.patcog.2026.114038_b46","series-title":"Mot20: A benchmark for multi object tracking in crowded scenes","author":"Dendorfer","year":"2020"},{"key":"10.1016\/j.patcog.2026.114038_b47","doi-asserted-by":"crossref","unstructured":"X. Wang, L. Xie, C. Dong, Y. Shan, Real-esrgan: Training real-world blind super-resolution with pure synthetic data, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 1905\u20131914.","DOI":"10.1109\/ICCVW54120.2021.00217"},{"key":"10.1016\/j.patcog.2026.114038_b48","series-title":"A higher order metric for evaluating multi-object tracking","first-page":"548","author":"Luiten","year":"2021"},{"key":"10.1016\/j.patcog.2026.114038_b49","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1155\/2008\/246309","article-title":"Evaluating multiple object tracking performance: the clear mot metrics","volume":"2008","author":"Bernardin","year":"2008","journal-title":"EURASIP J. Image Video Process."},{"key":"10.1016\/j.patcog.2026.114038_b50","series-title":"European Conference on Computer Vision","first-page":"17","article-title":"Performance measures and a data set for multi-target, multi-camera tracking","author":"Ristani","year":"2016"},{"key":"10.1016\/j.patcog.2026.114038_b51","doi-asserted-by":"crossref","unstructured":"K. He, X. Zhang, S. Ren, J. Sun, Deep residual learning for image recognition, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2016, pp. 770\u2013778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"10.1016\/j.patcog.2026.114038_b52","unstructured":"W. Luo, Y. Zhong, Y. Gan, L. Ma, et al., CO-MOT: Boosting End-to-end Transformer-based Multi-Object Tracking via Coopetition Label Assignment and Shadow Sets, in: The Thirteenth International Conference on Learning Representations, 2025."},{"key":"10.1016\/j.patcog.2026.114038_b53","doi-asserted-by":"crossref","unstructured":"R. Gao, L. Wang, MeMOTR: Long-term memory-augmented transformer for multi-object tracking, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 9901\u20139910.","DOI":"10.1109\/ICCV51070.2023.00908"},{"key":"10.1016\/j.patcog.2026.114038_b54","first-page":"6504","article-title":"Hybrid-sort: Weak cues matter for online multi-object tracking","volume":"vol. 38","author":"Yang","year":"2024"},{"key":"10.1016\/j.patcog.2026.114038_b55","doi-asserted-by":"crossref","unstructured":"S.W. Zamir, A. Arora, S. Khan, M. Hayat, F.S. Khan, M.-H. Yang, Restormer: Efficient transformer for high-resolution image restoration, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 5728\u20135739.","DOI":"10.1109\/CVPR52688.2022.00564"}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326010034?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326010034?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T19:43:22Z","timestamp":1783194202000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0031320326010034"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,12]]},"references-count":55,"alternative-id":["S0031320326010034"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.114038","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,12]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"VSE-MOT: Multi-object tracking in low-quality video scenes guided by visual semantic enhancement","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.114038","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"114038"}}