{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,14]],"date-time":"2026-05-14T18:16:06Z","timestamp":1778782566636,"version":"3.51.4"},"reference-count":46,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2023YFC3306304"],"award-info":[{"award-number":["2023YFC3306304"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100016109","name":"Taishan Industry Leading Talents","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100016109","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100014103","name":"Key Technology Research and Development Program of Shandong","doi-asserted-by":"publisher","award":["2024TSGC0031"],"award-info":[{"award-number":["2024TSGC0031"]}],"id":[{"id":"10.13039\/100014103","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1016\/j.eswa.2026.131881","type":"journal-article","created":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T11:42:00Z","timestamp":1772797320000},"page":"131881","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["MSMD: Robust pedestrian detection via uncertainty-aware multi-scale fusion"],"prefix":"10.1016","volume":"317","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-3120-9248","authenticated-orcid":false,"given":"Zhixin","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-1158-8740","authenticated-orcid":false,"given":"Zhongquan","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-5429-0946","authenticated-orcid":false,"given":"Youwei","family":"Qiao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0268-0217","authenticated-orcid":false,"given":"Xiaoming","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1146-0833","authenticated-orcid":false,"given":"Xiangzhi","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3304-6294","authenticated-orcid":false,"given":"Yanling","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.131881_bib0001","series-title":"European conference on computer vision","first-page":"213","article-title":"End-to-end object detection with transformers","author":"Carion","year":"2020"},{"key":"10.1016\/j.eswa.2026.131881_bib0002","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"9650","article-title":"Emerging properties in self-supervised vision transformers","author":"Caron","year":"2021"},{"key":"10.1016\/j.eswa.2026.131881_bib0003","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.111209","article-title":"BiFPN-YOLO: One-stage object detection integrating bi-directional feature pyramid networks","volume":"160","author":"Doherty","year":"2025","journal-title":"Pattern Recognition"},{"key":"10.1016\/j.eswa.2026.131881_bib0004","series-title":"2023 IEEE 3rd international maghreb meeting of the conference on sciences and techniques of automatic control and computer engineering (MI-STA)","first-page":"171","article-title":"Smart surveillance system using deep learning","author":"El-Shekhi","year":"2023"},{"key":"10.1016\/j.eswa.2026.131881_bib0005","series-title":"Proceedings of the 7th international conference on frontiers of educational technologies, ICFET \u201921","first-page":"144","article-title":"Deep learning in smart video surveillance for crowd management: A systematic literature review","author":"Garcia","year":"2021"},{"key":"10.1016\/j.eswa.2026.131881_bib0006","series-title":"2020 25th international conference on pattern recognition (ICPR)","first-page":"2288","article-title":"Teacher-student training and triplet loss for facial expression recognition under occlusion","author":"Georgescu","year":"2021"},{"key":"10.1016\/j.eswa.2026.131881_bib0007","unstructured":"Gu, A., & Dao, T. (2023). Mamba: Linear-time sequence modeling with selective state spaces. arXiv: 2312.00752."},{"key":"10.1016\/j.eswa.2026.131881_bib0008","unstructured":"Gu, A., Goel, K., & R\u00e9, C. (2021). Efficiently modeling long sequences with structured state spaces. arXiv: 2111.00396."},{"key":"10.1016\/j.eswa.2026.131881_bib0009","doi-asserted-by":"crossref","first-page":"78","DOI":"10.1016\/j.patrec.2023.08.019","article-title":"Towards accurate dense pedestrian detection via occlusion-prediction aware label assignment and hierarchical-NMS","volume":"174","author":"He","year":"2023","journal-title":"Pattern Recognition Letters"},{"key":"10.1016\/j.eswa.2026.131881_bib0010","unstructured":"Hinton, G., Vinyals, O., & Dean, J. (2015). Distilling the knowledge in a neural network. arXiv: 1503.02531."},{"key":"10.1016\/j.eswa.2026.131881_bib0011","doi-asserted-by":"crossref","first-page":"401","DOI":"10.1016\/j.inffus.2023.02.014","article-title":"Multimodal pedestrian detection using metaheuristics with deep convolutional neural network in crowded scenes","volume":"95","author":"Jain","year":"2023","journal-title":"Information Fusion"},{"key":"10.1016\/j.eswa.2026.131881_bib0012","unstructured":"Jocher, G., Chaurasia, A., Qiu, J., (2023). Ultralytics yolov8. https:\/\/github.com\/ultralytics\/ultralytics."},{"key":"10.1016\/j.eswa.2026.131881_bib0013","article-title":"ultralytics\/yolov5: v3. 0","author":"Jocher","year":"2020","journal-title":"Zenodo"},{"key":"10.1016\/j.eswa.2026.131881_bib0014","unstructured":"Khanam, R., & Hussain, M. (2024). Yolov11: An overview of the key architectural enhancements. arXiv: 2410.17725."},{"key":"10.1016\/j.eswa.2026.131881_bib0015","doi-asserted-by":"crossref","unstructured":"Kim, S., Xiao, R., Georgescu, M.-I., Alaniz, S., & Akata, Z. (2024). Cosmos: Cross-modality self-distillation for vision language pre-training. arXiv: 2412.01814.","DOI":"10.1109\/CVPR52734.2025.01369"},{"key":"10.1016\/j.eswa.2026.131881_bib0016","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2024.129199","article-title":"Self-supervised visual learning in the low-data regime: A comparative evaluation","volume":"620","author":"Konstantakos","year":"2025","journal-title":"Neurocomputing"},{"key":"10.1016\/j.eswa.2026.131881_bib0017","doi-asserted-by":"crossref","unstructured":"Lee, S., Choi, J., & Kim, H. J. (2024). EfficientVIM: Efficient vision mamba with hidden state mixer based state space duality. arXiv: 2411.15241.","DOI":"10.1109\/CVPR52734.2025.01390"},{"key":"10.1016\/j.eswa.2026.131881_bib0018","series-title":"2023 IEEE\/CVF conference on computer vision and pattern recognition (CVPR)","first-page":"10439","article-title":"Blind video deflickering by neural filtering with a flawed atlas","author":"Lei","year":"2023"},{"key":"10.1016\/j.eswa.2026.131881_bib0019","series-title":"2024 international conference on image processing, computer vision and machine learning (ICICML)","first-page":"1012","article-title":"Vm-unet++: Advanced nested vision mamba unet for precise medical image segmentation","author":"Lei","year":"2024"},{"key":"10.1016\/j.eswa.2026.131881_bib0020","unstructured":"Li, C., Li, L., Jiang, H., Weng, K., Geng, Y., Li, L., Ke, Z., Li, Q., Cheng, M., Nie, W. et al. (2022). Yolov6: A single-stage object detection framework for industrial applications. arXiv: 2209.02976."},{"key":"10.1016\/j.eswa.2026.131881_bib0021","series-title":"2023 IEEE\/CVF conference on computer vision and pattern recognition (CVPR)","first-page":"20156","article-title":"Rethinking feature-based knowledge distillation for face recognition","author":"Li","year":"2023"},{"key":"10.1016\/j.eswa.2026.131881_bib0022","unstructured":"Li, T., Li, C., Lyu, J., Pei, H., Zhang, B., Jin, T., & Ji, R. (2025). Damamba: Vision state space model with dynamic adaptive scan. arXiv: 2502.12627."},{"key":"10.1016\/j.eswa.2026.131881_bib0023","series-title":"Computer vision\u2013ECCV 2014: 13th european conference, Zurich, Switzerland, September 6-12, 2014, proceedings, Part V 13","first-page":"740","article-title":"Microsoft coco: Common objects in context","author":"Lin","year":"2014"},{"key":"10.1016\/j.eswa.2026.131881_bib0024","doi-asserted-by":"crossref","first-page":"754","DOI":"10.1109\/TIP.2020.3038371","article-title":"Coupled network for robust pedestrian detection with gated multi-layer feature extraction and deformable occlusion handling","volume":"30","author":"Liu","year":"2021","journal-title":"IEEE Transactions on Image Processing"},{"key":"10.1016\/j.eswa.2026.131881_bib0025","first-page":"103031","article-title":"Vmamba: Visual state space model","volume":"37","author":"Liu","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"202","key":"10.1016\/j.eswa.2026.131881_bib0026","doi-asserted-by":"crossref","first-page":"202","DOI":"10.1007\/s10462-025-11206-w","article-title":"An enhanced framework for real-time dense crowd abnormal behavior detection using YOLOv8","volume":"58","author":"Nasir","year":"2025","journal-title":"Artificial Intelligence Review"},{"issue":"1","key":"10.1016\/j.eswa.2026.131881_bib0027","doi-asserted-by":"crossref","first-page":"577","DOI":"10.1007\/s40747-020-00206-8","article-title":"Survey of pedestrian detection with occlusion","volume":"7","author":"Ning","year":"2021","journal-title":"Complex & Intelligent Systems"},{"key":"10.1016\/j.eswa.2026.131881_bib0028","unstructured":"Oquab, M., Darcet, T., Moutakanni, T., Vo, H., Szafraniec, M., Khalidov, V., Fernandez, P., Haziza, D., Massa, F., El-Nouby, A. et al. (2023). Dinov2: Learning robust visual features without supervision. arXiv: 2304.07193."},{"key":"10.1016\/j.eswa.2026.131881_bib0029","series-title":"Technical Report","article-title":"Global status report on road safety 2023","author":"Organization","year":"2023"},{"key":"10.1016\/j.eswa.2026.131881_bib0030","series-title":"2012 IEEE conference on computer vision and pattern recognition","first-page":"3258","article-title":"A discriminative deep model for pedestrian detection with occlusion handling","author":"Ouyang","year":"2012"},{"key":"10.1016\/j.eswa.2026.131881_bib0031","unstructured":"Peng, Y., Li, H., Wu, P., Zhang, Y., Sun, X., & Wu, F. (2024). D-FINE: redefine regression task in DETRs as fine-grained distribution refinement. arXiv: 2410.13842."},{"key":"10.1016\/j.eswa.2026.131881_bib0032","unstructured":"Shao, S., Zhao, Z., Li, B., Xiao, T., Yu, G., Zhang, X., & Sun, J. (2018). Crowdhuman: A benchmark for detecting human in a crowd. arXiv: 1805.00123."},{"key":"10.1016\/j.eswa.2026.131881_bib0033","doi-asserted-by":"crossref","first-page":"8483","DOI":"10.1109\/TIP.2021.3115672","article-title":"Autopedestrian: An automatic data augmentation and loss function search scheme for pedestrian detection","volume":"30","author":"Tang","year":"2021","journal-title":"IEEE Transactions on Image Processing"},{"key":"10.1016\/j.eswa.2026.131881_bib0034","first-page":"107984","article-title":"Yolov10: Real-time end-to-end object detection","volume":"37","author":"Wang","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"8","key":"10.1016\/j.eswa.2026.131881_bib0035","doi-asserted-by":"crossref","first-page":"8205","DOI":"10.1609\/aaai.v39i8.32885","article-title":"Mamba YOLO: A simple baseline for object detection with state space model","volume":"39","author":"Wang","year":"2025","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"10.1016\/j.eswa.2026.131881_bib0036","doi-asserted-by":"crossref","DOI":"10.1016\/j.measurement.2023.113442","article-title":"Infrared pedestrian detection using improved UNet and YOLO through sharing visible light domain information","volume":"221","author":"Wei","year":"2023","journal-title":"Measurement"},{"key":"10.1016\/j.eswa.2026.131881_bib0037","series-title":"Computer vision \u2013 ECCV 2022","first-page":"516","article-title":"Self-filtering: A noise-aware sample selection for label noise with confidence penalization","author":"Wei","year":"2022"},{"key":"10.1016\/j.eswa.2026.131881_bib0038","series-title":"2023 IEEE\/CVF international conference on computer vision (ICCV)","first-page":"14895","article-title":"Sefd: Learning to distill complex pose and occlusion","author":"Yang","year":"2023"},{"key":"10.1016\/j.eswa.2026.131881_bib0039","unstructured":"Zhan, Z., Kong, Z., Gong, Y., Wu, Y., Meng, Z., Zheng, H., Shen, X., Ioannidis, S., Niu, W., Zhao, P. et al. (2024). Exploring token pruning in vision state space models. arXiv: 2409.18962."},{"key":"10.1016\/j.eswa.2026.131881_bib0040","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"3213","article-title":"Citypersons: A diverse dataset for pedestrian detection","author":"Zhang","year":"2017"},{"issue":"2","key":"10.1016\/j.eswa.2026.131881_bib0041","doi-asserted-by":"crossref","first-page":"380","DOI":"10.1109\/TMM.2019.2929005","article-title":"Widerperson: A diverse dataset for dense pedestrian detection in the wild","volume":"22","author":"Zhang","year":"2020","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.eswa.2026.131881_bib0042","series-title":"Proceedings of the 37th international conference on machine learning, Proceedings of Machine Learning Research","first-page":"11163","article-title":"Dual-path distillation: A unified framework to improve black-box attacks","volume":"vol. 119","author":"Zhang","year":"2020"},{"key":"10.1016\/j.eswa.2026.131881_bib0043","doi-asserted-by":"crossref","DOI":"10.1016\/j.imavis.2024.105365","article-title":"Information gap based knowledge distillation for occluded facial expression recognition","volume":"154","author":"Zhang","year":"2025","journal-title":"Image and Vision Computing"},{"key":"10.1016\/j.eswa.2026.131881_bib0044","series-title":"2024 IEEE\/CVF conference on computer vision and pattern recognition (CVPR)","first-page":"16965","article-title":"Detrs beat yolos on real-time object detection","author":"Zhao","year":"2024"},{"key":"10.1016\/j.eswa.2026.131881_bib0045","series-title":"Proceedings of the 41st international conference on machine learning","first-page":"62429","article-title":"Vision mamba: Efficient visual representation learning with bidirectional state space model","volume":"vol. 235","author":"Zhu","year":"2024"},{"key":"10.1016\/j.eswa.2026.131881_bib0046","unstructured":"Zhu, X., Su, W., Lu, L., Li, B., Wang, X., & Dai, J. (2020). Deformable detr: Deformable transformers for end-to-end object detection. arXiv: 2010.04159."}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426007943?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426007943?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,14]],"date-time":"2026-05-14T17:51:59Z","timestamp":1778781119000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426007943"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6]]},"references-count":46,"alternative-id":["S0957417426007943"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.131881","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,6]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"MSMD: Robust pedestrian detection via uncertainty-aware multi-scale fusion","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.131881","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"131881"}}