{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T06:36:35Z","timestamp":1782974195343,"version":"3.54.5"},"reference-count":52,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1016\/j.eswa.2026.132308","type":"journal-article","created":{"date-parts":[[2026,4,14]],"date-time":"2026-04-14T05:33:49Z","timestamp":1776144829000},"page":"132308","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":1,"special_numbering":"C","title":["Efficient dual-modality object detection with state-space fusion and Mix attention"],"prefix":"10.1016","volume":"323","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-5060-2157","authenticated-orcid":false,"given":"Huanyu","family":"Yang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shijie","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3870-2361","authenticated-orcid":false,"given":"Jun","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0228-6749","authenticated-orcid":false,"given":"Yuming","family":"Bo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4176-3947","authenticated-orcid":false,"given":"Jiacun","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4039-891X","authenticated-orcid":false,"given":"Giancarlo","family":"Fortino","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.132308_bib0001","doi-asserted-by":"crossref","DOI":"10.1016\/j.compag.2024.109090","article-title":"Agricultural object detection with you only look once (YOLO) algorithm: A bibliometric and systematic literature review","volume":"223","author":"Badgujar","year":"2024","journal-title":"Computers and Electronics in Agriculture"},{"key":"10.1016\/j.eswa.2026.132308_bib0002","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"403","article-title":"Multimodal object detection by channel switching and spatial attention","author":"Cao","year":"2023"},{"key":"10.1016\/j.eswa.2026.132308_bib0003","series-title":"International conference on pattern recognition","first-page":"236","article-title":"DEYOLO: Dual-feature-enhancement YOLO for cross-modality object detection","author":"Chen","year":"2024"},{"key":"10.1016\/j.eswa.2026.132308_bib0004","doi-asserted-by":"crossref","first-page":"6971","DOI":"10.1109\/TMM.2022.3216476","article-title":"Does thermal really always matter for RGB-T salient object detection?","volume":"25","author":"Cong","year":"2022","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.eswa.2026.132308_bib0005","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision (ICCV)","first-page":"2988","article-title":"Dynamic DETR: End-to-end object detection with dynamic attention","author":"Dai","year":"2021"},{"key":"10.1016\/j.eswa.2026.132308_bib0006","doi-asserted-by":"crossref","DOI":"10.1109\/TMM.2025.3599020","article-title":"Fusion-Mamba for cross-modality object detection","author":"Dong","year":"2025","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.eswa.2026.132308_bib0007","first-page":"127181","article-title":"Demystify Mamba in vision: A linear attention perspective","volume":"37","author":"Han","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132308_bib0008","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"13713","article-title":"Coordinate attention for efficient mobile network design","author":"Hou","year":"2021"},{"key":"10.1016\/j.eswa.2026.132308_bib0009","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"3496","article-title":"LLVIP: A visible-infrared paired dataset for low-light vision","author":"Jia","year":"2021"},{"key":"10.1016\/j.eswa.2026.132308_bib0010","doi-asserted-by":"crossref","first-page":"1066","DOI":"10.1016\/j.procs.2022.01.135","article-title":"A review of YOLO algorithm developments","volume":"199","author":"Jiang","year":"2022","journal-title":"Procedia Computer Science"},{"key":"10.1016\/j.eswa.2026.132308_bib0011","doi-asserted-by":"crossref","first-page":"185","DOI":"10.1016\/j.inffus.2022.09.019","article-title":"Current advances and future perspectives of image fusion: A comprehensive review","volume":"90","author":"Karim","year":"2023","journal-title":"Information Fusion"},{"key":"10.1016\/j.eswa.2026.132308_bib0012","unstructured":"Khanam, R., & Hussain, M. (2024). YOLOv11: An overview of the key architectural enhancements. arXiv: 2410.17725\">arXiv preprint arXiv: 2410.17725. 10.48550\/arXiv.2410.17725."},{"key":"10.1016\/j.eswa.2026.132308_bib0013","unstructured":"Li, C., Song, D., Tong, R., & Tang, M. (2018a). Multispectral pedestrian detection via simultaneous detection and segmentation. arXiv: 1808.04818\">arXiv preprint arXiv: 1808.04818. 10.48550\/arXiv.1808.04818."},{"issue":"10","key":"10.1016\/j.eswa.2026.132308_bib0014","doi-asserted-by":"crossref","first-page":"2913","DOI":"10.1109\/TCSVT.2018.2874312","article-title":"Learning local-global multi-graph descriptors for RGB-T object tracking","volume":"29","author":"Li","year":"2018","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.eswa.2026.132308_bib0015","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR)","first-page":"13619","article-title":"DN-DETR: Accelerate DETR training by introducing query denoising","author":"Li","year":"2022"},{"key":"10.1016\/j.eswa.2026.132308_bib0016","doi-asserted-by":"crossref","DOI":"10.1109\/TCSVT.2025.3587918","article-title":"CFMW: Cross-modality fusion Mamba for robust object detection under adverse weather","author":"Li","year":"2025","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.eswa.2026.132308_bib0017","doi-asserted-by":"crossref","first-page":"852","DOI":"10.1109\/TMM.2023.3272471","article-title":"Multiscale cross-modal homogeneity enhancement and confidence-aware fusion for multispectral pedestrian detection","volume":"26","author":"Li","year":"2023","journal-title":"IEEE Transactions on Multimedia"},{"issue":"1","key":"10.1016\/j.eswa.2026.132308_bib0018","doi-asserted-by":"crossref","DOI":"10.1007\/s11432-019-2899-9","article-title":"AR-CNN: An attention ranking network for learning urban perception","volume":"65","author":"Li","year":"2022","journal-title":"Science China Information Sciences"},{"key":"10.1016\/j.eswa.2026.132308_bib0019","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"5802","article-title":"Target-aware dual adversarial learning and a multi-scenario multi-modality benchmark to fuse infrared and visible for object detection","author":"Liu","year":"2022"},{"issue":"10","key":"10.1016\/j.eswa.2026.132308_bib0020","doi-asserted-by":"crossref","first-page":"7226","DOI":"10.1109\/TCSVT.2022.3168999","article-title":"Revisiting modality-specific feature compensation for visible-infrared person re-identification","volume":"32","author":"Liu","year":"2022","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.eswa.2026.132308_bib0021","series-title":"European conference on computer vision","first-page":"21","article-title":"Ssd: Single shot multibox detector","author":"Liu","year":"2016"},{"issue":"7","key":"10.1016\/j.eswa.2026.132308_bib0022","doi-asserted-by":"crossref","first-page":"1200","DOI":"10.1109\/JAS.2022.105686","article-title":"Swinfusion: Cross-domain long-range learning for general image fusion via swin transformer","volume":"9","author":"Ma","year":"2022","journal-title":"IEEE\/CAA Journal of Automatica Sinica"},{"issue":"2","key":"10.1016\/j.eswa.2026.132308_bib0023","doi-asserted-by":"crossref","first-page":"599","DOI":"10.3390\/s23020599","article-title":"Infrared and visible image fusion technology and application: A review","volume":"23","author":"Ma","year":"2023","journal-title":"Sensors"},{"key":"10.1016\/j.eswa.2026.132308_bib0024","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"6443","article-title":"Efficientvmamba: Atrous selective scan for light weight visual Mamba","volume":"vol. 39","author":"Pei","year":"2025"},{"key":"10.1016\/j.eswa.2026.132308_bib0025","article-title":"Cross-modality fusion transformer for multispectral object detection","author":"Qingyun","year":"2021","journal-title":"SSRN Electronic Journal"},{"key":"10.1016\/j.eswa.2026.132308_bib0026","unstructured":"Rahman, M. M., Tutul, A. A., Nath, A., Laishram, L., Jung, S. K., & Hammond, T. (2024). Mamba in vision: A comprehensive survey of techniques and applications. arXiv: 2410.03105\">arXiv preprint arXiv: 2410.03105. 10.48550\/arXiv.2410.03105."},{"key":"10.1016\/j.eswa.2026.132308_bib0027","series-title":"Proceedings of the 30th ACM international conference on multimedia","first-page":"4003","article-title":"Detfusion: A detection-driven infrared and visible image fusion network","author":"Sun","year":"2022"},{"key":"10.1016\/j.eswa.2026.132308_bib0028","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2023.101870","article-title":"Rethinking the necessity of image fusion in high-level vision tasks: A practical infrared and visible image fusion network based on progressive semantic injection and scene fidelity","volume":"99","author":"Tang","year":"2023","journal-title":"Information Fusion"},{"key":"10.1016\/j.eswa.2026.132308_bib0029","series-title":"European conference on computer vision","first-page":"1","article-title":"YOLOv9: Learning what you want to learn using programmable gradient information","author":"Wang","year":"2024"},{"key":"10.1016\/j.eswa.2026.132308_bib0030","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"11534","article-title":"ECA-Net: Efficient channel attention for deep convolutional neural networks","author":"Wang","year":"2020"},{"issue":"8","key":"10.1016\/j.eswa.2026.132308_bib0031","doi-asserted-by":"crossref","first-page":"2122","DOI":"10.1007\/s11263-023-01784-z","article-title":"Multi-modal 3D object detection in autonomous driving: A survey","volume":"131","author":"Wang","year":"2023","journal-title":"International Journal of Computer Vision"},{"key":"10.1016\/j.eswa.2026.132308_bib0032","series-title":"Proceedings of the European conference on computer vision (ECCV)","first-page":"3","article-title":"CBAM: Convolutional block attention module","author":"Woo","year":"2018"},{"issue":"1","key":"10.1016\/j.eswa.2026.132308_bib0033","doi-asserted-by":"crossref","first-page":"502","DOI":"10.1109\/TPAMI.2020.3012548","article-title":"U2Fusion: A unified unsupervised image fusion network","volume":"44","author":"Xu","year":"2020","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.132308_bib0034","unstructured":"Xu, R., Yang, S., Wang, Y., Cai, Y., Du, B., & Chen, H. (2024). Visual Mamba: A survey and new outlooks. arXiv: 2404.18861\">arXiv preprint arXiv: 2404.18861. 10.48550\/arXiv.2404.18861."},{"key":"10.1016\/j.eswa.2026.132308_bib0035","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2024.129289","article-title":"ISTD-DETR: A deep learning algorithm based on DETR and super-resolution for infrared small target detection","volume":"621","author":"Yang","year":"2025","journal-title":"Neurocomputing"},{"issue":"4","key":"10.1016\/j.eswa.2026.132308_bib0036","doi-asserted-by":"crossref","first-page":"599","DOI":"10.1109\/THMS.2025.3566102","article-title":"DMPD: A dual-modality fusion method for cross-spectral pedestrian detection","volume":"55","author":"Yang","year":"2025","journal-title":"IEEE Transactions on Human-Machine Systems"},{"issue":"23","key":"10.1016\/j.eswa.2026.132308_bib0037","doi-asserted-by":"crossref","first-page":"5527","DOI":"10.3390\/rs15235527","article-title":"Efficient detection of forest fire smoke in UAV aerial imagery based on an improved YOLOv5 model and transfer learning","volume":"15","author":"Yang","year":"2023","journal-title":"Remote Sensing"},{"key":"10.1016\/j.eswa.2026.132308_bib0038","unstructured":"Yao, J., Hong, D., Li, C., & Chanussot, J. (2024). Spectralmamba: Efficient Mamba for hyperspectral image classification. arXiv: 2404.08489\">arXiv preprint arXiv: 2404.08489. 10.48550\/arXiv.2404.08489."},{"key":"10.1016\/j.eswa.2026.132308_bib0039","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2024.102246","article-title":"Improving RGB-infrared object detection with cascade alignment-guided transformer","volume":"105","author":"Yuan","year":"2024","journal-title":"Information Fusion"},{"key":"10.1016\/j.eswa.2026.132308_bib0040","series-title":"2020\u202fIEEE international conference on image processing (ICIP)","first-page":"276","article-title":"Multispectral fusion for object detection with cyclic fuse-and-refine blocks","author":"Zhang","year":"2020"},{"key":"10.1016\/j.eswa.2026.132308_bib0041","series-title":"Proceedings of the IEEE\/CVF winter conference on applications of computer vision","first-page":"72","article-title":"Guided attentive feature fusion for multispectral pedestrian detection","author":"Zhang","year":"2021"},{"issue":"13","key":"10.1016\/j.eswa.2026.132308_bib0042","doi-asserted-by":"crossref","first-page":"5683","DOI":"10.3390\/app14135683","article-title":"A survey on visual Mamba","volume":"14","author":"Zhang","year":"2024","journal-title":"Applied Sciences"},{"key":"10.1016\/j.eswa.2026.132308_bib0043","first-page":"1","article-title":"VLF-DETR: Integrating vision-language and high-frequency features for transmission line defect detection","volume":"74","author":"Zhang","year":"2025","journal-title":"IEEE Transactions on Instrumentation and Measurement"},{"issue":"11","key":"10.1016\/j.eswa.2026.132308_bib0044","doi-asserted-by":"crossref","first-page":"6510","DOI":"10.1109\/TSMC.2024.3386873","article-title":"Transmission line component defect detection based on UAV patrol images: A self-supervised HC-ViT method","volume":"54","author":"Zhang","year":"2024","journal-title":"IEEE Transactions on Systems, Man, and Cybernetics: Systems"},{"key":"10.1016\/j.eswa.2026.132308_bib0045","article-title":"Weakly aligned feature fusion for multimodal object detection","author":"Zhang","year":"2021","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"issue":"8","key":"10.1016\/j.eswa.2026.132308_bib0046","doi-asserted-by":"crossref","first-page":"10535","DOI":"10.1109\/TPAMI.2023.3261282","article-title":"Visible and infrared image fusion using deep learning","volume":"45","author":"Zhang","year":"2023","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.132308_bib0047","article-title":"Removal then selection: A coarse-to-fine fusion perspective for RGB-infrared object detection","author":"Zhao","year":"2025","journal-title":"IEEE Transactions on Intelligent Transportation Systems"},{"key":"10.1016\/j.eswa.2026.132308_bib0048","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"5906","article-title":"CDDFuse: Correlation-driven dual-branch feature decomposition for multi-modality image fusion","author":"Zhao","year":"2023"},{"key":"10.1016\/j.eswa.2026.132308_bib0049","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.112103","article-title":"LESOD: Lightweight and efficient network for RGB-D salient object detection","volume":"171","author":"Zhong","year":"2026","journal-title":"Pattern Recognition"},{"key":"10.1016\/j.eswa.2026.132308_bib0050","series-title":"Proceedings ofthe IEEE\/CVF conference on computer vision and pattern recognition","first-page":"13065","article-title":"Squeeze-and-attention networks for semantic segmentation","author":"Zhong","year":"2020"},{"key":"10.1016\/j.eswa.2026.132308_bib0051","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2023.122256","article-title":"A YOLO-NL object detector for real-time detection","volume":"238","author":"Zhou","year":"2024","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132308_bib0052","series-title":"Vision Mamba: Efficient visual representation learning with bidirectional state space model","author":"Zhu","year":"2024"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426012212?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426012212?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,17]],"date-time":"2026-06-17T23:25:02Z","timestamp":1781738702000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426012212"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8]]},"references-count":52,"alternative-id":["S0957417426012212"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132308","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,8]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Efficient dual-modality object detection with state-space fusion and Mix attention","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132308","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"132308"}}