{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,13]],"date-time":"2026-06-13T00:56:10Z","timestamp":1781312170863,"version":"3.54.1"},"reference-count":63,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62476220"],"award-info":[{"award-number":["62476220"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100021171","name":"Basic and Applied Basic Research Foundation of Guangdong Province","doi-asserted-by":"publisher","award":["2024A1515030186"],"award-info":[{"award-number":["2024A1515030186"]}],"id":[{"id":"10.13039\/501100021171","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2023YFC3209304"],"award-info":[{"award-number":["2023YFC3209304"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100017596","name":"Natural Science Basic Research Program of Shaanxi Province","doi-asserted-by":"publisher","award":["2024JC-DXWT-07"],"award-info":[{"award-number":["2024JC-DXWT-07"]}],"id":[{"id":"10.13039\/501100017596","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1016\/j.eswa.2026.132380","type":"journal-article","created":{"date-parts":[[2026,4,7]],"date-time":"2026-04-07T15:24:19Z","timestamp":1775575459000},"page":"132380","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["CDFNet: Cross-dimension fusion network with dual feature enhancement for multimodal object detection"],"prefix":"10.1016","volume":"322","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9067-9667","authenticated-orcid":false,"given":"Wencong","family":"Wu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7230-1476","authenticated-orcid":false,"given":"Xiuwei","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1086-2873","authenticated-orcid":false,"given":"Hanlin","family":"Yin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-4257-5282","authenticated-orcid":false,"given":"Haorui","family":"Zeng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-8954-9991","authenticated-orcid":false,"given":"Chenxu","family":"Wei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9106-7515","authenticated-orcid":false,"given":"Lei","family":"Yu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2977-8057","authenticated-orcid":false,"given":"Yanning","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.132380_bib0001","series-title":"IEEE\/CVF conference on computer vision and pattern recognition","first-page":"2997","article-title":"DaFF: Dual attentive feature fusion for multispectral pedestrian detection","author":"Althoupety","year":"2024"},{"key":"10.1016\/j.eswa.2026.132380_bib0002","series-title":"Ieee international conference on image processing","first-page":"2620","article-title":"Multimodal transformer using cross-channel attention for object detection in remote sensing images","author":"Bahaduri","year":"2024"},{"key":"10.1016\/j.eswa.2026.132380_bib0003","series-title":"IEEE\/CVF conference on computer vision and pattern recognition","first-page":"403","article-title":"Multimodal object detection by channel switching and spatial attention","author":"Cao","year":"2023"},{"key":"10.1016\/j.eswa.2026.132380_bib0004","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2023.110423","article-title":"IGT: illumination-guided RGB-T object detection with transformers","volume":"268","author":"Chen","year":"2023","journal-title":"Knowledge-Based Systems"},{"issue":"24","key":"10.1016\/j.eswa.2026.132380_bib0005","doi-asserted-by":"crossref","first-page":"65603","DOI":"10.1007\/s11042-023-17949-4","article-title":"A review of object detection: Datasets, performance evaluation, architecture, applications and current trends","volume":"83","author":"Chen","year":"2024","journal-title":"Multimedia Tools and Applications"},{"key":"10.1016\/j.eswa.2026.132380_bib0006","series-title":"European conference on computer vision","first-page":"139","article-title":"Multimodal object detection via probabilistic ensembling","volume":"vol. 13669","author":"Chen","year":"2022"},{"issue":"5","key":"10.1016\/j.eswa.2026.132380_bib0007","doi-asserted-by":"crossref","first-page":"4713","DOI":"10.1109\/TCSVT.2024.3520318","article-title":"SeaDATE: Remedy dual-attention transformer with semantic alignment via contrast learning for multimodal object detection","volume":"35","author":"Dong","year":"2025","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.eswa.2026.132380_bib0008","doi-asserted-by":"crossref","first-page":"7392","DOI":"10.1109\/TMM.2025.3599020","article-title":"Fusion-mamba for cross-modality object detection","volume":"27","author":"Dong","year":"2025","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.eswa.2026.132380_bib0009","series-title":"International conference on learning representations","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2021"},{"key":"10.1016\/j.eswa.2026.132380_bib0010","doi-asserted-by":"crossref","unstructured":"Fang, Q., Han, D., & Wang, Z. (2021). Cross-modality fusion transformer for multispectral object detection. arXiv: 2111.00273.","DOI":"10.2139\/ssrn.4227745"},{"key":"10.1016\/j.eswa.2026.132380_bib0011","article-title":"Cross-modality attentive feature fusion for object detection in multispectral remote sensing imagery","volume":"130","author":"Fang","year":"2022","journal-title":"Pattern Recognition"},{"key":"10.1016\/j.eswa.2026.132380_bib0012","unstructured":"FLIR (2018). Flir thermal dataset for algorithm training. https:\/\/www.flir.in\/oem\/adas\/adas-dataset-form."},{"issue":"10","key":"10.1016\/j.eswa.2026.132380_bib0013","doi-asserted-by":"crossref","first-page":"13232","DOI":"10.1109\/TNNLS.2023.3266452","article-title":"LRAF-Net: Long-range attention fusion network for visible-infrared object detection","volume":"35","author":"Fu","year":"2024","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"10.1016\/j.eswa.2026.132380_bib0014","doi-asserted-by":"crossref","first-page":"148","DOI":"10.1016\/j.inffus.2018.11.017","article-title":"Fusion of multispectral data through illumination-aware deep neural networks for pedestrian detection","volume":"50","author":"Guan","year":"2019","journal-title":"Information Fusion"},{"issue":"7","key":"10.1016\/j.eswa.2026.132380_bib0015","doi-asserted-by":"crossref","first-page":"7101","DOI":"10.1109\/TCSVT.2025.3539625","article-title":"Ei2det: Edge-guided illumination-aware interactive learning for visible-infrared object detection","volume":"35","author":"Hu","year":"2025","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.eswa.2026.132380_bib0016","series-title":"European conference on artificial intelligence","first-page":"482","article-title":"SFDFusion: An efficient spatial-frequency domain fusion network for infrared and visible image fusion","volume":"vol. 392","author":"Hu","year":"2024"},{"issue":"6","key":"10.1016\/j.eswa.2026.132380_bib0017","doi-asserted-by":"crossref","first-page":"6896","DOI":"10.1109\/TPAMI.2020.3007032","article-title":"Ccnet: Criss-cross attention for semantic segmentation","volume":"45","author":"Huang","year":"2023","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.132380_bib0018","series-title":"IEEE international conference on acoustics, speech and signal processing","first-page":"1","article-title":"CAMDET: Condition-adaptive multispectral object detection using a visible-thermal translation model","author":"Jang","year":"2025"},{"key":"10.1016\/j.eswa.2026.132380_bib0019","series-title":"IEEE\/CVF winter conference on applications of computer vision","first-page":"9437","article-title":"Multispectral object detection enhanced by cross-modal information complementary and cosine similarity channel resampling modules","author":"Jang","year":"2025"},{"key":"10.1016\/j.eswa.2026.132380_bib0020","series-title":"IEEE\/CVF international conference on computer vision workshops","first-page":"3489","article-title":"LLVIP: A visible-infrared paired dataset for low-light vision","author":"Jia","year":"2021"},{"key":"10.1016\/j.eswa.2026.132380_bib0021","unstructured":"Jocher, G. (2020). Yolov5 by ultralytics. https:\/\/github.com\/ultralytics\/yolov5."},{"key":"10.1016\/j.eswa.2026.132380_bib0022","unstructured":"Jocher, G. (2024). ultralytics\/yolov8: v8.1.0-yolov8 oriented bounding boxes (obb). https:\/\/github.com\/ultralytics\/ultralytics."},{"key":"10.1016\/j.eswa.2026.132380_bib0023","series-title":"IEEE conference on computer vision and pattern recognition workshops","first-page":"243","article-title":"Fully convolutional region proposal networks for multispectral person detection","author":"K\u00f6nig","year":"2017"},{"key":"10.1016\/j.eswa.2026.132380_bib0024","doi-asserted-by":"crossref","first-page":"144","DOI":"10.1016\/j.patrec.2024.02.012","article-title":"Crossformer: Cross-guided attention for multi-modal object detection","volume":"179","author":"Lee","year":"2024","journal-title":"Pattern Recognition Letters"},{"key":"10.1016\/j.eswa.2026.132380_bib0025","series-title":"British machine vision conference","first-page":"225","article-title":"Multispectral pedestrian detection via simultaneous detection and segmentation","author":"Li","year":"2018"},{"issue":"7","key":"10.1016\/j.eswa.2026.132380_bib0026","doi-asserted-by":"crossref","first-page":"4060","DOI":"10.1109\/LRA.2023.3272269","article-title":"Explicit attention-enhanced fusion for RGB-thermal perception tasks","volume":"8","author":"Liang","year":"2023","journal-title":"IEEE Robotics and Automation Letters"},{"key":"10.1016\/j.eswa.2026.132380_bib0027","doi-asserted-by":"crossref","first-page":"158","DOI":"10.1016\/j.neucom.2022.07.054","article-title":"Polarized self-attention: Towards high-quality pixel-wise mapping","volume":"506","author":"Liu","year":"2022","journal-title":"Neurocomputing"},{"key":"10.1016\/j.eswa.2026.132380_bib0028","series-title":"IEEE\/CVF conference on computer vision and pattern recognition","first-page":"5792","article-title":"Target-aware dual adversarial learning and a multi-scenario multi-modality benchmark to fuse infrared and visible for object detection","author":"Liu","year":"2022"},{"issue":"8","key":"10.1016\/j.eswa.2026.132380_bib0029","doi-asserted-by":"crossref","first-page":"9467","DOI":"10.1109\/TITS.2024.3360875","article-title":"Yolo-3DMM for simultaneous multiple object detection and tracking in traffic scenarios","volume":"25","author":"Liu","year":"2024","journal-title":"IEEE Transactions on Intelligent Transportation Systems (T-ITS)"},{"key":"10.1016\/j.eswa.2026.132380_bib0030","series-title":"IEEE\/CVF international conference on computer vision","first-page":"9992","article-title":"Swin transformer: Hierarchical vision transformer using shifted windows","author":"Liu","year":"2021"},{"key":"10.1016\/j.eswa.2026.132380_bib0031","series-title":"IEEE winter conference on applications of computer vision","first-page":"3138","article-title":"Rotate to attend: Convolutional triplet attention module","author":"Misra","year":"2021"},{"key":"10.1016\/j.eswa.2026.132380_bib0032","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2024.126359","article-title":"SC3D: semantic-guided and class-adaptive cross-domain fusion for 3D object detection in autonomous vehicles","volume":"268","author":"Mushtaq","year":"2025","journal-title":"Expert Systems with Applications"},{"issue":"6","key":"10.1016\/j.eswa.2026.132380_bib0033","doi-asserted-by":"crossref","first-page":"1137","DOI":"10.1109\/TPAMI.2016.2577031","article-title":"Faster R-CNN: Towards real-time object detection with region proposal networks","volume":"39","author":"Ren","year":"2017","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.132380_bib0034","doi-asserted-by":"crossref","first-page":"26","DOI":"10.1016\/j.patrec.2024.05.001","article-title":"MOD-YOLO: multispectral object detection based on transformer dual-stream YOLO","volume":"183","author":"Shao","year":"2024","journal-title":"Pattern Recognition Letters"},{"key":"10.1016\/j.eswa.2026.132380_bib0035","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2023.109913","article-title":"ICAFusion: Iterative cross-attention guided feature fusion for multispectral object detection","volume":"145","author":"Shen","year":"2024","journal-title":"Pattern Recognition"},{"issue":"11","key":"10.1016\/j.eswa.2026.132380_bib0036","doi-asserted-by":"crossref","first-page":"15407","DOI":"10.1109\/TITS.2024.3439557","article-title":"Robustness-aware 3D object detection in autonomous driving: A review and outlook","volume":"25","author":"Song","year":"2024","journal-title":"IEEE Transactions on Intelligent Transportation Systems"},{"issue":"10","key":"10.1016\/j.eswa.2026.132380_bib0037","doi-asserted-by":"crossref","first-page":"6700","DOI":"10.1109\/TCSVT.2022.3168279","article-title":"Drone-based RGB-infrared cross-modality vehicle detection via uncertainty-aware learning","volume":"32","author":"Sun","year":"2022","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.eswa.2026.132380_bib0038","doi-asserted-by":"crossref","first-page":"79","DOI":"10.1016\/j.inffus.2022.03.007","article-title":"Piafusion: A progressive infrared and visible image fusion network based on illumination aware","volume":"83-84","author":"Tang","year":"2022","journal-title":"Information Fusion"},{"key":"10.1016\/j.eswa.2026.132380_bib0039","unstructured":"Ultralytics (2024). Ultralytics YOLOv11. https:\/\/docs.ultralytics.com\/models\/yolo11."},{"key":"10.1016\/j.eswa.2026.132380_bib0040","series-title":"Advances in neural information processing systems","first-page":"5998","article-title":"Attention is all you need","author":"Vaswani","year":"2017"},{"key":"10.1016\/j.eswa.2026.132380_bib0041","series-title":"IEEE\/CVF conference on computer vision and pattern recognition","first-page":"7464","article-title":"Yolov7: Trainable bag-of-freebies sets new state-of-the-art for real-time object detectors","author":"Wang","year":"2023"},{"key":"10.1016\/j.eswa.2026.132380_bib0042","first-page":"1","article-title":"Inducing causal meta-knowledge from virtual domain: Causal meta-generalization for hyperspectral domain generalization","volume":"62","author":"Wang","year":"2024","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.eswa.2026.132380_bib0043","first-page":"1","article-title":"Kcdnet: Multimodal object detection in modal information imbalance scenes","volume":"73","author":"Wang","year":"2024","journal-title":"IEEE Transactions on Instrumentation and Measurement"},{"key":"10.1016\/j.eswa.2026.132380_bib0044","first-page":"1","article-title":"Cross-modal oriented object detection of UAV aerial images based on image feature","volume":"62","author":"Wang","year":"2024","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.eswa.2026.132380_bib0045","series-title":"International conference on pattern recognition","first-page":"284","article-title":"RGB-T object detection via group shuffled multi-receptive attention and multi-modal supervision","volume":"vol. 15317","author":"Wang","year":"2024"},{"key":"10.1016\/j.eswa.2026.132380_bib0046","series-title":"IEEE\/CVF conference on computer vision and pattern recognition","first-page":"17201","article-title":"Depth-aware concealed crop detection in dense agricultural scenes","author":"Wang","year":"2024"},{"key":"10.1016\/j.eswa.2026.132380_bib0047","doi-asserted-by":"crossref","first-page":"8166","DOI":"10.1109\/JSTARS.2023.3294624","article-title":"Vehicle detection based on adaptive multimodal feature fusion and cross-modal vehicle index using RGB-T images","volume":"16","author":"Wu","year":"2023","journal-title":"IEEE Journal of Selected Topics in Applied Earth Observations and Remote Sensing"},{"issue":"4","key":"10.1016\/j.eswa.2026.132380_bib0048","doi-asserted-by":"crossref","first-page":"2132","DOI":"10.1109\/TCDS.2023.3238181","article-title":"YOLO-MS: Multispectral object detection via feature interaction and self-attention guided fusion","volume":"15","author":"Xie","year":"2023","journal-title":"IEEE Transactions on Cognitive and Developmental Systems"},{"issue":"1","key":"10.1016\/j.eswa.2026.132380_bib0049","doi-asserted-by":"crossref","first-page":"547","DOI":"10.1109\/TCSVT.2024.3454631","article-title":"Multidimensional fusion network for multispectral object detection","volume":"35","author":"Yang","year":"2025","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.eswa.2026.132380_bib0050","doi-asserted-by":"crossref","first-page":"1172","DOI":"10.1109\/LSP.2023.3309578","article-title":"Multi-scale aggregation transformers for multispectral object detection","volume":"30","author":"You","year":"2023","journal-title":"IEEE Signal Processing Letters"},{"key":"10.1016\/j.eswa.2026.132380_bib0051","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2024.102246","article-title":"Improving RGB-infrared object detection with cascade alignment-guided transformer","volume":"105","author":"Yuan","year":"2024","journal-title":"Information Fusion"},{"key":"10.1016\/j.eswa.2026.132380_bib0052","series-title":"European conference on computer vision","first-page":"509","article-title":"Translation, scale and rotation: Cross-modal alignment meets RGB-infrared vehicle detection","volume":"vol. 13669","author":"Yuan","year":"2022"},{"issue":"11","key":"10.1016\/j.eswa.2026.132380_bib0053","doi-asserted-by":"crossref","first-page":"11198","DOI":"10.1109\/TCSVT.2024.3418965","article-title":"MMI-DET: Exploring multi-modal integration for visible and infrared object detection","volume":"34","author":"Zeng","year":"2024","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.eswa.2026.132380_bib0054","series-title":"IEEE international conference on image processing","first-page":"276","article-title":"Multispectral fusion for object detection with cyclic fuse-and-refine blocks","author":"Zhang","year":"2020"},{"key":"10.1016\/j.eswa.2026.132380_bib0055","series-title":"IEEE winter conference on applications of computer vision","first-page":"72","article-title":"Guided attentive feature fusion for multispectral pedestrian detection","author":"Zhang","year":"2021"},{"key":"10.1016\/j.eswa.2026.132380_bib0056","first-page":"1","article-title":"SuperYOLO: Super resolution assisted object detection in multimodal remote sensing imagery","volume":"61","author":"Zhang","year":"2023","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.eswa.2026.132380_bib0057","doi-asserted-by":"crossref","first-page":"20","DOI":"10.1016\/j.inffus.2018.09.015","article-title":"Cross-modality interactive attention network for multispectral pedestrian detection","volume":"50","author":"Zhang","year":"2019","journal-title":"Information Fusion"},{"key":"10.1016\/j.eswa.2026.132380_bib0058","first-page":"1","article-title":"Illumination-guided RGBT object detection with inter- and intra-modality fusion","volume":"72","author":"Zhang","year":"2023","journal-title":"IEEE Transactions on Instrumentation and Measurement"},{"key":"10.1016\/j.eswa.2026.132380_bib0059","first-page":"1","article-title":"Removal then selection: A coarse-to-fine fusion perspective for RGB-infrared object detection","author":"Zhao","year":"2025","journal-title":"IEEE Transactions on Intelligent Transportation Systems"},{"key":"10.1016\/j.eswa.2026.132380_bib0060","series-title":"IEEE\/CVF conference on computer vision and pattern recognition","first-page":"13955","article-title":"Metafusion: Infrared and visible image fusion via meta-feature embedding from object detection","author":"Zhao","year":"2023"},{"key":"10.1016\/j.eswa.2026.132380_bib0061","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2024.125826","article-title":"Differential multimodal fusion algorithm for remote sensing object detection through multi-branch feature extraction","volume":"265","author":"Zhao","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132380_bib0062","series-title":"Aaai conference on artificial intelligence","first-page":"12993","article-title":"Distance-IOU loss: Faster and better learning for bounding box regression","author":"Zheng","year":"2020"},{"key":"10.1016\/j.eswa.2026.132380_bib0063","series-title":"European conference on computer vision","first-page":"787","article-title":"Improving multispectral pedestrian detection by addressing modality imbalance problems","volume":"vol. 12363","author":"Zhou","year":"2020"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426012935?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426012935?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,13]],"date-time":"2026-06-13T00:04:42Z","timestamp":1781309082000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426012935"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8]]},"references-count":63,"alternative-id":["S0957417426012935"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132380","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,8]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"CDFNet: Cross-dimension fusion network with dual feature enhancement for multimodal object detection","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132380","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"132380"}}