{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T12:03:39Z","timestamp":1784203419325,"version":"3.55.0"},"reference-count":56,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62373102"],"award-info":[{"award-number":["62373102"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Neural Networks"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.neunet.2026.108988","type":"journal-article","created":{"date-parts":[[2026,4,17]],"date-time":"2026-04-17T15:45:42Z","timestamp":1776440742000},"page":"108988","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["MDM: Modality decoupling for visible and infrared Mamba-based object detection"],"prefix":"10.1016","volume":"202","author":[{"given":"Yucheng","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3960-8828","authenticated-orcid":false,"given":"Lin","family":"Chai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.neunet.2026.108988_bib0001","doi-asserted-by":"crossref","DOI":"10.1016\/j.imavis.2024.105387","article-title":"HSIRMamba: An effective feature learning for hyperspectral image classification using residual Mamba","volume":"154","author":"Arya","year":"2025","journal-title":"Image and Vision Computing"},{"key":"10.1016\/j.neunet.2026.108988_bib0002","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2025.114051","article-title":"A front-back view fusion strategy and a novel dataset for super tiny object detection in remote sensing imagery","volume":"326","author":"Bai","year":"2025","journal-title":"Knowledge-Based Systems"},{"key":"10.1016\/j.neunet.2026.108988_bib0004","series-title":"2017 IEEE conference on computer vision and pattern recognition (CVPR)","first-page":"95","article-title":"Unsupervised pixel-level domain adaptation with generative adversarial networks","author":"Bousmalis","year":"2017"},{"key":"10.1016\/j.neunet.2026.108988_bib0005","series-title":"2023 IEEE\/CVF conference on computer vision and pattern recognition workshops (CVPRW)","first-page":"403","article-title":"Multimodal object detection by channel switching and spatial attention","author":"Cao","year":"2023"},{"key":"10.1016\/j.neunet.2026.108988_bib0006","doi-asserted-by":"crossref","DOI":"10.1016\/j.cviu.2025.104436","article-title":"UM-Mamba: An efficient U-network with medical visual state space for medical image segmentation","volume":"259","author":"Chen","year":"2025","journal-title":"Computer Vision and Image Understanding"},{"key":"10.1016\/j.neunet.2026.108988_bib0007","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2025.130505","article-title":"Alignment-assisted Frequency Fusion Network for RGB-infrared vehicle detection","volume":"647","author":"Chen","year":"2025","journal-title":"Neurocomputing"},{"key":"10.1016\/j.neunet.2026.108988_bib0008","first-page":"1","article-title":"Object detection in autonomous driving scenario using YOLOv8-SimAM: a robust test in different datasets","author":"Cheng","year":"2025","journal-title":"Transportmetrica A: Transport Science"},{"key":"10.1016\/j.neunet.2026.108988_bib0009","unstructured":"Dong, W. et al., (2024). Fusion-Mamba for cross-modality object detection. arXiv.Org. https:\/\/arxiv.org\/abs\/2404.09146."},{"key":"10.1016\/j.neunet.2026.108988_bib0003","series-title":"In International Conference on Learning Representations","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2021"},{"key":"10.1016\/j.neunet.2026.108988_bib0010","article-title":"Cross-modality fusion transformer for multispectral object detection","author":"Fang","year":"2022","journal-title":"SSRN Electronic Journal"},{"issue":"10","key":"10.1016\/j.neunet.2026.108988_bib0011","doi-asserted-by":"crossref","first-page":"13232","DOI":"10.1109\/TNNLS.2023.3266452","article-title":"LRAF-Net: Long-range attention fusion network for visible\u2013infrared object detection","volume":"35","author":"Fu","year":"2024","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"10.1016\/j.neunet.2026.108988_bib0012","unstructured":"Gu, A., & Dao, T. (2023). Mamba: Linear-time sequence modeling with selective state spaces. arXiv.Org. https:\/\/arxiv.org\/abs\/2312.00752."},{"issue":"17","key":"10.1016\/j.neunet.2026.108988_bib0013","doi-asserted-by":"crossref","first-page":"17243","DOI":"10.1109\/JSEN.2022.3186889","article-title":"YOLOX-SAR: High-precision object detection system based on visible and infrared sensors for SAR remote sensing","volume":"22","author":"Guo","year":"2022","journal-title":"IEEE Sensors Journal"},{"key":"10.1016\/j.neunet.2026.108988_bib0014","series-title":"Proceedings of the 31st ACM international conference on multimedia","first-page":"1465","article-title":"Multispectral object detection via cross-modal conflict-aware learning","author":"He","year":"2023"},{"key":"10.1016\/j.neunet.2026.108988_bib0015","doi-asserted-by":"crossref","DOI":"10.1016\/j.infrared.2023.105107","article-title":"An object detection algorithm based on infrared-visible dual modal feature fusion","volume":"137","author":"Hou","year":"2024","journal-title":"Infrared Physics & Technology"},{"key":"10.1016\/j.neunet.2026.108988_bib0016","series-title":"2021 IEEE\/CVF international conference on computer vision workshops (ICCVW)","first-page":"3489","article-title":"LLVIP: A visible-infrared paired dataset for low-light vision","author":"Jia","year":"2021"},{"key":"10.1016\/j.neunet.2026.108988_bib0017","doi-asserted-by":"crossref","first-page":"144","DOI":"10.1016\/j.patrec.2024.02.012","article-title":"CrossFormer: Cross-guided attention for multi-modal object detection","volume":"179","author":"Lee","year":"2024","journal-title":"Pattern Recognition Letters"},{"key":"10.1016\/j.neunet.2026.108988_bib0018","unstructured":"Li, H. et al., (2024). CFMW: Cross-modality fusion Mamba for robust object detection under adverse weather. arXiv.Org. https:\/\/arxiv.org\/abs\/2404.16302."},{"key":"10.1016\/j.neunet.2026.108988_bib0019","doi-asserted-by":"crossref","DOI":"10.1016\/j.infrared.2025.105851","article-title":"DFLMF-ISTD: Infrared small object detection network based on decoupled feature learning and multi-scale feature fusion","volume":"149","author":"Li","year":"2025","journal-title":"Infrared Physics & Technology"},{"key":"10.1016\/j.neunet.2026.108988_bib0020","article-title":"YOLOSR-IST: A deep learning network for small target detection in infrared remote sensing images based on super-resolution And&nbsp;YOLO","author":"Li","year":"2022","journal-title":"SSRN Electronic Journal"},{"key":"10.1016\/j.neunet.2026.108988_bib0021","doi-asserted-by":"crossref","first-page":"8678","DOI":"10.1109\/TMM.2024.3381377","article-title":"M2FNet: Mask-guided multi-level fusion for RGB-T pedestrian detection","volume":"26","author":"Li","year":"2024","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.neunet.2026.108988_bib0022","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2025.103414","article-title":"COMO: Cross-Mamba interaction and offset-guided fusion for multimodal object detection","volume":"125","author":"Liu","year":"2026","journal-title":"Information Fusion"},{"key":"10.1016\/j.neunet.2026.108988_bib0023","series-title":"2022 IEEE\/CVF conference on computer vision and pattern recognition (CVPR)","first-page":"5792","article-title":"Target-aware dual adversarial learning and a multi-scenario multi-modality benchmark to fuse infrared and visible for object detection","author":"Liu","year":"2022"},{"key":"10.1016\/j.neunet.2026.108988_bib0024","doi-asserted-by":"crossref","DOI":"10.1016\/j.infrared.2025.105985","article-title":"Infrared and visible image fusion based on spatial correlation attention","volume":"150","author":"Liu","year":"2025","journal-title":"Infrared Physics & Technology"},{"key":"10.1016\/j.neunet.2026.108988_bib0025","unstructured":"Liu, Y. et al., (2024). VMamba: Visual state space model. arXiv.Org. https:\/\/arxiv.org\/abs\/2401.10166."},{"issue":"2","key":"10.1016\/j.neunet.2026.108988_bib0026","doi-asserted-by":"crossref","first-page":"1825","DOI":"10.1109\/TCSVT.2024.3484761","article-title":"Exploring relational knowledge for source-free domain adaptation","volume":"35","author":"Ma","year":"2025","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.neunet.2026.108988_bib0027","doi-asserted-by":"crossref","DOI":"10.1016\/j.autcon.2025.106139","article-title":"Integrating text parsing and object detection for automated monitoring of finishing works in construction projects","volume":"174","author":"Oh","year":"2025","journal-title":"Automation in Construction"},{"key":"10.1016\/j.neunet.2026.108988_bib0028","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2025.129913","article-title":"DACFusion: Dual asymmetric cross-attention guided feature fusion for multispectral object detection","volume":"635","author":"Qian","year":"2025","journal-title":"Neurocomputing"},{"key":"10.1016\/j.neunet.2026.108988_bib0029","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2022.108786","article-title":"Cross-modality attentive feature fusion for object detection in multispectral remote sensing imagery","volume":"130","author":"Qingyun","year":"2022","journal-title":"Pattern Recognition"},{"key":"10.1016\/j.neunet.2026.108988_bib0030","doi-asserted-by":"crossref","DOI":"10.1016\/j.imavis.2022.104518","article-title":"An improved YOLOv5 method for large objects detection with multi-scale feature cross-layer fusion network","volume":"125","author":"Qu","year":"2022","journal-title":"Image and Vision Computing"},{"key":"10.1016\/j.neunet.2026.108988_bib0031","doi-asserted-by":"crossref","first-page":"187","DOI":"10.1016\/j.jvcir.2015.11.002","article-title":"Vehicle detection in aerial imagery : A small target detection benchmark","volume":"34","author":"Razakarivony","year":"2016","journal-title":"Journal of Visual Communication and Image Representation"},{"key":"10.1016\/j.neunet.2026.108988_bib0032","unstructured":"Ren, S. et al., (2015). Faster R-CNN: Towards real-time object detection with region proposal networks. arXiv.Org. https:\/\/arxiv.org\/abs\/1506.01497."},{"key":"10.1016\/j.neunet.2026.108988_bib0033","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2023.109913","article-title":"ICAFusion: Iterative cross-attention guided feature fusion for multispectral object detection","volume":"145","author":"Shen","year":"2024","journal-title":"Pattern Recognition"},{"key":"10.1016\/j.neunet.2026.108988_bib0034","series-title":"Smart Methane Emission Detection System Development Final Report","author":"Spidle","year":"2021"},{"issue":"10","key":"10.1016\/j.neunet.2026.108988_bib0035","doi-asserted-by":"crossref","first-page":"6700","DOI":"10.1109\/TCSVT.2022.3168279","article-title":"Drone-based RGB-infrared cross-modality vehicle detection via uncertainty-aware learning","volume":"32","author":"Sun","year":"2022","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.neunet.2026.108988_bib0036","doi-asserted-by":"crossref","first-page":"79","DOI":"10.1016\/j.inffus.2022.03.007","article-title":"PIAFusion: A progressive infrared and visible image fusion network based on illumination aware","volume":"83\u201384","author":"Tang","year":"2022","journal-title":"Information Fusion"},{"key":"10.1016\/j.neunet.2026.108988_bib0037","first-page":"1","article-title":"Cross-modal oriented object detection of UAV aerial images based on image feature","volume":"62","author":"Wang","year":"2024","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.108988_bib0038","series-title":"Proceedings of the 31st ACM International Conference on Multimedia","first-page":"2663","article-title":"TIRDet: Mono-modality thermal infrared object detection based on prior thermal-to-visible translation","author":"Wang","year":"2023"},{"key":"10.1016\/j.neunet.2026.108988_bib0039","series-title":"2024 IEEE\/CVF conference on computer vision and pattern recognition workshops (CVPRW)","first-page":"5541","article-title":"GM-DETR: Generalized muiltispectral detection transformer with efficient fusion encoder for visible-infrared detection","author":"Xiao","year":"2024"},{"issue":"1","key":"10.1016\/j.neunet.2026.108988_bib0040","doi-asserted-by":"crossref","first-page":"547","DOI":"10.1109\/TCSVT.2024.3454631","article-title":"Multidimensional fusion network for multispectral object detection","volume":"35","author":"Yang","year":"2025","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.neunet.2026.108988_bib0041","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2025.103007","article-title":"Deep learning based infrared small object segmentation: Challenges and future directions","volume":"118","author":"Yang","year":"2025","journal-title":"Information Fusion"},{"key":"10.1016\/j.neunet.2026.108988_bib0042","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2024.102246","article-title":"Improving RGB-infrared object detection with cascade alignment-guided transformer","volume":"105","author":"Yuan","year":"2024","journal-title":"Information Fusion"},{"key":"10.1016\/j.neunet.2026.108988_bib0043","series-title":"Proceedings of the 33rd ACM international conference on multimedia","first-page":"2409","article-title":"UniRGB-IR: A unified framework for visible-infrared semantic tasks via adapter tuning","author":"Yuan","year":"2025"},{"key":"10.1016\/j.neunet.2026.108988_bib0044","series-title":"Lecture notes in computer science","doi-asserted-by":"crossref","first-page":"509","DOI":"10.1007\/978-3-031-20077-9_30","article-title":"Translation, scale and rotation: Cross-modal alignment meets RGB-infrared vehicle detection","author":"Yuan","year":"2022"},{"key":"10.1016\/j.neunet.2026.108988_bib0045","first-page":"1","article-title":"C2Former: Calibrated and complementary transformer for RGB-infrared object detection","volume":"62","author":"Yuan","year":"2024","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"issue":"11","key":"10.1016\/j.neunet.2026.108988_bib0046","doi-asserted-by":"crossref","first-page":"11198","DOI":"10.1109\/TCSVT.2024.3418965","article-title":"MMI-Det: Exploring multi-modal integration for visible and infrared object detection","volume":"34","author":"Zeng","year":"2024","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.neunet.2026.108988_bib0047","series-title":"2020 IEEE international conference on image processing (ICIP)","first-page":"276","article-title":"Multispectral Fusion for Object Detection with Cyclic Fuse-and-Refine Blocks","author":"Zhang","year":"2020"},{"key":"10.1016\/j.neunet.2026.108988_bib0048","doi-asserted-by":"crossref","DOI":"10.1016\/j.infrared.2025.105895","article-title":"TMCN: Text-guided Mamba-CNN dual-encoder network for infrared and visible image fusion","volume":"149","author":"Zhang","year":"2025","journal-title":"Infrared Physics & Technology"},{"issue":"7","key":"10.1016\/j.neunet.2026.108988_bib0049","doi-asserted-by":"crossref","first-page":"13276","DOI":"10.1109\/TNNLS.2024.3443455","article-title":"TFDet: Target-aware fusion for RGB-T pedestrian detection","volume":"36","author":"Zhang","year":"2025","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"issue":"6","key":"10.1016\/j.neunet.2026.108988_bib0050","doi-asserted-by":"crossref","first-page":"3728","DOI":"10.1109\/TIV.2024.3462488","article-title":"Rethinking early-fusion strategies for improved multispectral object detection","volume":"10","author":"Zhang","year":"2025","journal-title":"IEEE Transactions on Intelligent Vehicles"},{"issue":"7","key":"10.1016\/j.neunet.2026.108988_bib0051","doi-asserted-by":"crossref","DOI":"10.1007\/s10489-025-06470-w","article-title":"MKDFusion: Modality knowledge decoupled for infrared and visible image fusion","volume":"55","author":"Zhang","year":"2025","journal-title":"Applied Intelligence"},{"issue":"2","key":"10.1016\/j.neunet.2026.108988_bib0052","doi-asserted-by":"crossref","first-page":"2504","DOI":"10.1109\/TITS.2025.3638627","article-title":"Removal Then Selection: A Coarse-to-Fine Fusion Perspective for RGB-Infrared Object Detection","volume":"27","author":"Zhao","year":"2026","journal-title":"IEEE Transactions on Intelligent Transportation Systems"},{"key":"10.1016\/j.neunet.2026.108988_bib0053","series-title":"2023 IEEE\/CVF conference on computer vision and pattern recognition (CVPR)","first-page":"5906","article-title":"CDDFuse: Correlation-driven dual-branch feature decomposition for multi-modality image fusion","author":"Zhao","year":"2023"},{"key":"10.1016\/j.neunet.2026.108988_bib0054","first-page":"1","article-title":"Reflectance-guided progressive feature alignment network for all-day UAV object detection","volume":"63","author":"Zhao","year":"2025","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.108988_bib0055","first-page":"1","article-title":"DMM: Disparity-guided multispectral Mamba for oriented object detection in remote sensing","volume":"63","author":"Zhou","year":"2025","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.108988_bib0056","unstructured":"Zhu, L. et al., (2024). Vision Mamba: Efficient visual representation learning with bidirectional state space model. arXiv.Org. https:\/\/arxiv.org\/abs\/2401.09417."}],"container-title":["Neural Networks"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0893608026004491?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0893608026004491?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T11:11:19Z","timestamp":1784200279000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0893608026004491"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":56,"alternative-id":["S0893608026004491"],"URL":"https:\/\/doi.org\/10.1016\/j.neunet.2026.108988","relation":{},"ISSN":["0893-6080"],"issn-type":[{"value":"0893-6080","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"MDM: Modality decoupling for visible and infrared Mamba-based object detection","name":"articletitle","label":"Article Title"},{"value":"Neural Networks","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.neunet.2026.108988","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"108988"}}