{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T23:11:52Z","timestamp":1778109112556,"version":"3.51.4"},"reference-count":49,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100007129","name":"Shandong Province Natural Science Foundation","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100007129","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Digital Signal Processing"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1016\/j.dsp.2026.106094","type":"journal-article","created":{"date-parts":[[2026,3,25]],"date-time":"2026-03-25T18:56:27Z","timestamp":1774464987000},"page":"106094","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["A Cross-Modal hierarchical enhanced fusion method for object detection in intelligent transportation systems"],"prefix":"10.1016","volume":"177","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-0419-9582","authenticated-orcid":false,"given":"Lihui","family":"Lu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sihao","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuke","family":"Gu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bo","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bowen","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.dsp.2026.106094_bib0001","doi-asserted-by":"crossref","first-page":"1773","DOI":"10.1109\/TITS.2013.2266661","article-title":"Looking at vehicles on the road: a survey of vision-Based vehicle detection, tracking, and behavior analysis","volume":"14","author":"Sivaraman","year":"2013","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"10.1016\/j.dsp.2026.106094_bib0002","doi-asserted-by":"crossref","first-page":"3234","DOI":"10.1109\/TITS.2020.2993926","article-title":"Deep neural network based vehicle and pedestrian detection for autonomous driving: a survey","volume":"22","author":"Chen","year":"2021","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"10.1016\/j.dsp.2026.106094_bib0003","doi-asserted-by":"crossref","first-page":"1459","DOI":"10.1109\/TIP.2004.836169","article-title":"Statistical modeling of complex backgrounds for foreground object detection","volume":"13","author":"Li","year":"2004","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.dsp.2026.106094_bib0004","series-title":"Iberoamerican Congress on Pattern Recognition","first-page":"59","article-title":"Bus detection for intelligent transport systems using computer vision","volume":"8259","author":"Gerschuni","year":"2013"},{"key":"10.1016\/j.dsp.2026.106094_bib0005","doi-asserted-by":"crossref","first-page":"8591","DOI":"10.3390\/s22228591","article-title":"Object relocation visual tracking based on histogram filter and siamese network in intelligent transportation","volume":"22","author":"Zhang","year":"2022","journal-title":"Sensors"},{"key":"10.1016\/j.dsp.2026.106094_bib0006","doi-asserted-by":"crossref","first-page":"15898","DOI":"10.1109\/TITS.2022.3146271","article-title":"ID-YOLO: Real-Time salient object detection based on the Driver\u2019s fixation region","volume":"23","author":"Qin","year":"2022","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"10.1016\/j.dsp.2026.106094_bib0007","doi-asserted-by":"crossref","first-page":"25345","DOI":"10.1109\/TITS.2022.3158253","article-title":"Edge YOLO: real-Time intelligent object detection system based on edge-Cloud cooperation in autonomous vehicles","volume":"23","author":"Liang","year":"2022","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"10.1016\/j.dsp.2026.106094_bib0008","doi-asserted-by":"crossref","DOI":"10.1038\/s41598-024-82356-0","article-title":"Multi-sensor fusion and segmentation for autonomous vehicle multi-object tracking using deep q networks","volume":"14","author":"Vinoth","year":"2024","journal-title":"Sci. Rep."},{"key":"10.1016\/j.dsp.2026.106094_bib0009","series-title":"2014IEEE Conference on Computer Vision and Pattern Recognition","first-page":"580","article-title":"Rich feature hierarchies for accurate object detection and semantic segmentation","author":"Girshick","year":"2014"},{"key":"10.1016\/j.dsp.2026.106094_bib0010","doi-asserted-by":"crossref","first-page":"1137","DOI":"10.1109\/TPAMI.2016.2577031","article-title":"Faster R-CNN: towards real-Time object detection with region proposal networks","volume":"39","author":"Ren","year":"2017","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.dsp.2026.106094_bib0011","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"779","article-title":"You only look once: unified, real-time object detection","author":"Redmon","year":"2016"},{"key":"10.1016\/j.dsp.2026.106094_bib0012","series-title":"European Conference on Computer Vision","first-page":"21","article-title":"Ssd: single shot multibox detector","author":"Liu","year":"2016"},{"key":"10.1016\/j.dsp.2026.106094_bib0013","series-title":"2017 IEEE International Conference on Computer Vision (ICCV)","first-page":"2999","article-title":"Focal loss for dense object detection","author":"Lin","year":"2017"},{"key":"10.1016\/j.dsp.2026.106094_bib0014","series-title":"2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"10778","article-title":"Efficientdet: scalable and efficient object detection","author":"Tan","year":"2020"},{"key":"10.1016\/j.dsp.2026.106094_bib0015","doi-asserted-by":"crossref","DOI":"10.1016\/j.imavis.2024.105276","article-title":"LVD-YOLO: An efficient lightweight vehicle detection model for intelligent transportation systems","volume":"151","author":"Pan","year":"2024","journal-title":"Image Vis. Comput."},{"key":"10.1016\/j.dsp.2026.106094_bib0016","doi-asserted-by":"crossref","first-page":"7649","DOI":"10.1038\/s41598-025-92148-9","article-title":"Study on lightweight strategies for L-YOLO algorithm in road object detection","volume":"15","author":"Hong","year":"2025","journal-title":"Sci. Rep."},{"key":"10.1016\/j.dsp.2026.106094_bib0017","doi-asserted-by":"crossref","DOI":"10.1016\/j.engappai.2024.109705","article-title":"The nexus of intelligent transportation: a lightweight bi-input fusion detection model for autonomous-rail rapid transit","volume":"139","author":"Tang","year":"2025","journal-title":"Eng. Appl. Artif. Intell."},{"key":"10.1016\/j.dsp.2026.106094_bib0018","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"16965","article-title":"Detrs beat yolos on real-time object detection","author":"Zhao","year":"2024"},{"key":"10.1016\/j.dsp.2026.106094_bib0019","doi-asserted-by":"crossref","first-page":"1735","DOI":"10.1109\/TMM.2021.3070138","article-title":"Deep-irtarget: an automatic target detector in infrared imagery using dual-domain feature extraction and allocation","volume":"24","author":"Zhang","year":"2022","journal-title":"IEEE Trans. Multimedia"},{"key":"10.1016\/j.dsp.2026.106094_bib0020","series-title":"Proceedings of the 41St International Conference on Machine Learning","article-title":"Vision mamba: efficient visual representation learning with bidirectional state space model","author":"Zhu","year":"2024"},{"key":"10.1016\/j.dsp.2026.106094_bib0021","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"12797","article-title":"Rethinking the image fusion: a fast unified image fusion network based on proportional maintenance of gradient and intensity","volume":"34","author":"Zhang","year":"2020"},{"key":"10.1016\/j.dsp.2026.106094_bib0022","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1007\/s10043-025-00977-w","article-title":"Adaptive cross-modal fusion for robust multi-modal object detection in infrared\u2013visible imaging","volume":"32","author":"Wu","year":"2025","journal-title":"Opt. Rev."},{"key":"10.1016\/j.dsp.2026.106094_bib0023","first-page":"1","article-title":"STDFusionNet: an infrared and visible image fusion network based on salient target detection","volume":"70","author":"Ma","year":"2021","journal-title":"IEEE Trans. Instrum. Meas."},{"key":"10.1016\/j.dsp.2026.106094_bib0024","doi-asserted-by":"crossref","first-page":"38","DOI":"10.1186\/s13634-023-01002-5","article-title":"Decision-level fusion detection method of visible and infrared images under low light conditions","volume":"2023","author":"Hu","year":"2023","journal-title":"EURASIP J. Adv. Signal Process."},{"key":"10.1016\/j.dsp.2026.106094_bib0025","article-title":"M2FNEt: multi-modal fusion network for object detection from visible and thermal infrared images","volume":"130","author":"Jiang","year":"2024","journal-title":"Int. J. Appl. Earth Obs. Geoinf."},{"key":"10.1016\/j.dsp.2026.106094_bib0026","doi-asserted-by":"crossref","DOI":"10.3389\/fphy.2024.1356248","article-title":"Cross-modality feature fusion for night pedestrian detection","volume":"12","author":"Feng","year":"2024","journal-title":"Front. Phys."},{"key":"10.1016\/j.dsp.2026.106094_bib0027","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1016\/j.aej.2024.09.012","article-title":"YOLO-Fusion And internet of things: advancing object detection in smart transportation","volume":"107","author":"Tang","year":"2024","journal-title":"Alexandria Eng.J."},{"key":"10.1016\/j.dsp.2026.106094_bib0028","doi-asserted-by":"crossref","first-page":"6735","DOI":"10.1109\/TCSVT.2023.3289142","article-title":"Differential feature awareness network within antagonistic learning for infrared-visible object detection","volume":"34","author":"Zhang","year":"2024","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.dsp.2026.106094_bib0029","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2023.109913","article-title":"ICAFusion: Iterative cross-attention guided feature fusion for multispectral object detection","volume":"145","author":"Shen","year":"2024","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.dsp.2026.106094_bib0030","first-page":"1","article-title":"C2former: calibrated and complementary transformer for rgb-infrared object detection","volume":"62","author":"Yuan","year":"2024","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.dsp.2026.106094_bib0031","series-title":"2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"11531","article-title":"ECA-Net: Efficient channel attention for deep convolutional neural networks","author":"Wang","year":"2020"},{"key":"10.1016\/j.dsp.2026.106094_bib0032","first-page":"1","article-title":"AGCA: An adaptive graph channel attention module for steel surface defect detection","volume":"72","author":"Xiang","year":"2023","journal-title":"IEEE Trans. Instrum. Meas."},{"key":"10.1016\/j.dsp.2026.106094_bib0033","series-title":"ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","first-page":"1","article-title":"Efficient multi-Scale attention module with cross-Spatial learning","author":"Ouyang","year":"2023"},{"key":"10.1016\/j.dsp.2026.106094_bib0034","doi-asserted-by":"crossref","unstructured":"W. Xu, Y. Wan, ELA: Efficient local attention for deep convolutional neural networks, arXiv preprint, arXiv: 2403.01123 22 (2024) 140. 10.1007\/s11554-025-01719-6.","DOI":"10.1007\/s11554-025-01719-6"},{"key":"10.1016\/j.dsp.2026.106094_bib0035","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"7132","article-title":"Squeeze-and-excitation networks","author":"Hu","year":"2018"},{"key":"10.1016\/j.dsp.2026.106094_bib0036","series-title":"Proceedings of the European Conference on Computer Vision (ECCV)","first-page":"3","article-title":"Cbam: convolutional block attention module","author":"Woo","year":"2018"},{"key":"10.1016\/j.dsp.2026.106094_bib0037","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"8205","article-title":"Mamba yolo: a simple baseline for object detection with state space model","volume":"39","author":"Wang","year":"2025"},{"key":"10.1016\/j.dsp.2026.106094_bib0038","doi-asserted-by":"crossref","first-page":"502","DOI":"10.1109\/TPAMI.2020.3012548","article-title":"U2Fusion: A unified unsupervised image fusion network","volume":"44","author":"Xu","year":"2022","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.dsp.2026.106094_bib0039","doi-asserted-by":"crossref","first-page":"5857","DOI":"10.3390\/app15115857","article-title":"EAD-PFNet: A multimodal fusion-Based multi-Scale road traffic detection algorithm","volume":"15","author":"Zhao","year":"2025","journal-title":"Appl. Sci."},{"key":"10.1016\/j.dsp.2026.106094_bib0040","doi-asserted-by":"crossref","unstructured":"J. Guo, C. Gao, F. Liu, D. Meng, X. Gao, DAMSDet: Dynamic adaptive multispectral detection transformer with competitive query selection and adaptive feature fusion(2024) 464\u2013481. 10.1007\/978-3-031-73383-3_27.","DOI":"10.1007\/978-3-031-73383-3_27"},{"key":"10.1016\/j.dsp.2026.106094_bib0041","doi-asserted-by":"crossref","first-page":"1200","DOI":"10.1109\/JAS.2022.105686","article-title":"Swinfusion: cross-domain long-range learning for general image fusion via swin transformer","volume":"9","author":"Ma","year":"2022","journal-title":"IEEE\/CAA J. Autom. Sin."},{"key":"10.1016\/j.dsp.2026.106094_bib0042","series-title":"Proceedings of the 30Th ACM International Conference on Multimedia","first-page":"4003","article-title":"Detfusion: a detection-driven infrared and visible image fusion network","author":"Sun","year":"2022"},{"key":"10.1016\/j.dsp.2026.106094_bib0043","doi-asserted-by":"crossref","DOI":"10.1016\/j.infrared.2025.106232","article-title":"Multimodal detection transformer with multiscale cross-modal feature fusion and selective query recollection","volume":"152","author":"Hou","year":"2026","journal-title":"Infrared Phys. Technol."},{"key":"10.1016\/j.dsp.2026.106094_bib0044","doi-asserted-by":"crossref","first-page":"7270","DOI":"10.3390\/s25237270","article-title":"A novel object detection algorithm combined YOLOv11 with dual-Encoder feature aggregation","volume":"25","author":"Chen","year":"2025","journal-title":"Sensors"},{"key":"10.1016\/j.dsp.2026.106094_bib0045","unstructured":"F. Qingyun, H. Dapeng, W. Zhaokui, Cross-modality fusion transformer for multispectral object detection, arXiv preprint, arXiv: 2111.00273(2021). 10.48550\/arXiv.2111.00273."},{"key":"10.1016\/j.dsp.2026.106094_bib0046","doi-asserted-by":"crossref","first-page":"26","DOI":"10.1016\/j.patrec.2024.05.001","article-title":"MOD-YOLO: Multispectral object detection based on transformer dual-stream YOLO","volume":"183","author":"Shao","year":"2024","journal-title":"Pattern Recognit. Lett."},{"key":"10.1016\/j.dsp.2026.106094_bib0047","first-page":"1","article-title":"Oriented infrared vehicle detection in aerial images via mining frequency and semantic information","volume":"61","author":"Zhang","year":"2023","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.dsp.2026.106094_bib0048","doi-asserted-by":"crossref","first-page":"364","DOI":"10.1109\/TIP.2022.3228497","article-title":"UIU-Net: U-Net In U-Net for infrared small object detection","volume":"32","author":"Wu","year":"2023","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.dsp.2026.106094_bib0049","first-page":"1","article-title":"DTNet: A specialized dual-Tuning network for infrared vehicle detection in aerial images","volume":"62","author":"Zhang","year":"2024","journal-title":"IEEE Trans. Geosci. Remote Sens."}],"container-title":["Digital Signal Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1051200426002137?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1051200426002137?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T22:26:18Z","timestamp":1778106378000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1051200426002137"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":49,"alternative-id":["S1051200426002137"],"URL":"https:\/\/doi.org\/10.1016\/j.dsp.2026.106094","relation":{},"ISSN":["1051-2004"],"issn-type":[{"value":"1051-2004","type":"print"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"A Cross-Modal hierarchical enhanced fusion method for object detection in intelligent transportation systems","name":"articletitle","label":"Article Title"},{"value":"Digital Signal Processing","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.dsp.2026.106094","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Inc. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"106094"}}