{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T04:07:25Z","timestamp":1784261245942,"version":"3.55.0"},"reference-count":44,"publisher":"Springer Science and Business Media LLC","issue":"13","license":[{"start":{"date-parts":[[2025,7,24]],"date-time":"2025-07-24T00:00:00Z","timestamp":1753315200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,7,24]],"date-time":"2025-07-24T00:00:00Z","timestamp":1753315200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2025,10]]},"DOI":"10.1007\/s00371-025-04071-9","type":"journal-article","created":{"date-parts":[[2025,7,24]],"date-time":"2025-07-24T13:50:28Z","timestamp":1753365028000},"page":"10827-10840","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["Cross-modal feature fusion via mutual assistance: a novel network for enhanced object detection"],"prefix":"10.1007","volume":"41","author":[{"given":"Xuebo","family":"Jin","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiaxi","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Huijun","family":"Ma","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tingli","family":"Su","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jianlei","family":"Kong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuting","family":"Bai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,7,24]]},"reference":[{"issue":"4","key":"4071_CR1","doi-asserted-by":"publisher","first-page":"280","DOI":"10.1016\/j.vrih.2023.06.002","volume":"6","author":"H Zhang","year":"2024","unstructured":"Zhang, H., et al.: MKEAH: multimodal knowledge extraction and accumulation based on hyperplane embedding for knowledge-based visual question answering. Virt. Real. Intell. Hardware 6(4), 280\u2013291 (2024)","journal-title":"Virt. Real. Intell. Hardware"},{"key":"4071_CR2","doi-asserted-by":"crossref","unstructured":"TMM-Nets: Transferred multi-to mono-modal generation for lupus retinopathy diagnosis. IEEE Transactions on Medical Imaging, 42(4), 1083-1094. (2022)","DOI":"10.1109\/TMI.2022.3223683"},{"key":"4071_CR3","doi-asserted-by":"crossref","unstructured":"Sthy-net: A feature fusion-enhanced dense-branched modules network for small thyroid nodule classification from ultrasound images. The Visual Computer, 39(8), 3675-3689. (2023)","DOI":"10.1007\/s00371-023-02984-x"},{"key":"4071_CR4","doi-asserted-by":"crossref","unstructured":"NHBS-Net: A feature fusion attention network for ultrasound neonatal hip bone segmentation. IEEE Transactions on Medical Imaging, 40(12), 3446-3458. (2021)","DOI":"10.1109\/TMI.2021.3087857"},{"key":"4071_CR5","doi-asserted-by":"crossref","unstructured":"Infrared image super-resolution method based on dual-branch deep neural network. The Visual Computer, 40(3), 1673-1684. (2024)","DOI":"10.1007\/s00371-023-02878-y"},{"key":"4071_CR6","doi-asserted-by":"crossref","unstructured":"Photohelper: Portrait photographing guidance via deep feature retrieval and fusion. IEEE Transactions on Multimedia, 25, 2226-2238. (2022)","DOI":"10.1109\/TMM.2022.3144890"},{"issue":"6","key":"4071_CR7","doi-asserted-by":"publisher","first-page":"509","DOI":"10.1016\/j.vrih.2023.06.003","volume":"5","author":"M Wang","year":"2023","unstructured":"Wang, M., et al.: Learning adequate alignment and interaction for cross-modal retrieval. Virt. Real. Intell. Hardware 5(6), 509\u2013522 (2023)","journal-title":"Virt. Real. Intell. Hardware"},{"issue":"5","key":"4071_CR8","doi-asserted-by":"publisher","first-page":"2434","DOI":"10.3390\/s23052434","volume":"23","author":"FS Neves","year":"2023","unstructured":"Neves, F.S., Claro, R.M., Pinto, A.M.: End-to-end detection of a landing platform for offshore UAVs based on a multimodal early fusion approach. Sensors 23(5), 2434 (2023)","journal-title":"Sensors"},{"issue":"8","key":"4071_CR9","first-page":"1332","volume":"9","author":"S Yan","year":"2007","unstructured":"Yan, S., Zhang, X., Kan, M.Y.: Early fusion of visual and audio features in event detection. IEEE Trans. Multimedia 9(8), 1332\u20131342 (2007)","journal-title":"IEEE Trans. Multimedia"},{"key":"4071_CR10","doi-asserted-by":"crossref","unstructured":"Snoek, C.G., Worring, M., Smeulders, A.W.: Early versus late fusion in semantic video analysis. In Proceedings of the 13th ACM international conference on multimedia (pp. 399-402). ACM. (2005)","DOI":"10.1145\/1101149.1101236"},{"key":"4071_CR11","doi-asserted-by":"crossref","unstructured":"Dhanaraj, M., Sharma, M., Sarkar, T., Karnam, S., Chachlakis, D.G., Ptucha, R., ... Saber, E.: Vehicle detection from multi-modal aerial imagery using YOLOv3 with mid-level fusion. In Big data II: learning, analytics, and applications (Vol. 11395, pp. 22-32) (2020)","DOI":"10.1117\/12.2558115"},{"issue":"6","key":"4071_CR12","doi-asserted-by":"publisher","first-page":"345","DOI":"10.1007\/s00530-010-0182-0","volume":"16","author":"PK Atrey","year":"2010","unstructured":"Atrey, P.K., Hossain, M.A., El Saddik, A., Kankanhalli, M.S.: Multimodal fusion for multimedia analysis: a survey. Multimedia Syst. 16(6), 345\u2013379 (2010)","journal-title":"Multimedia Syst."},{"key":"4071_CR13","unstructured":"Ngiam, J., Khosla, A., Kim, M., Nam, J., Lee, H., Ng, A.Y.: Multimodal deep learning. In Proceedings of the 28th international conference on machine learning (ICML-11) (pp. 689-696). (2011)"},{"key":"4071_CR14","unstructured":"Qingyun, F., Dapeng, H., Zhaokui, W.: Cross-modality fusion transformer for multispectral object detection. arXiv preprint arxiv: 2111.00273. (2021)"},{"key":"4071_CR15","doi-asserted-by":"crossref","unstructured":"Dong, W., Zhu, H., Lin, S., Luo, X., Shen, Y., Liu, X., ... Zhang, B.: Fusion-Mamba for Cross-modality Object Detection. arXiv e-prints, arxiv:2404. (2024)","DOI":"10.1109\/TMM.2025.3599020"},{"key":"4071_CR16","unstructured":"Jocher, G., Chaurasia, A., Qiu, J., Stoken, A.: YOLOv5: Implementation of YOLO (You Only Look Once) Version 5. Ultralytics. Available at: https:\/\/github.com\/ultralytics\/yolov5. (2020)"},{"key":"4071_CR17","doi-asserted-by":"publisher","first-page":"1158","DOI":"10.1109\/TMM.2023.3277281","volume":"26","author":"L Zhao","year":"2023","unstructured":"Zhao, L., Zhou, H., Zhu, X., Song, X., Li, H., Tao, W.: Lif-seg: Lidar and camera image fusion for 3D lidar semantic segmentation. IEEE Trans. Multimedia 26, 1158\u20131168 (2023)","journal-title":"IEEE Trans. Multimedia"},{"key":"4071_CR18","doi-asserted-by":"crossref","unstructured":"Vora, S., Lang, A.H., Helou, B., Beijbom, O.: Pointpainting: Sequential fusion for 3D object detection. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 4604-4612). (2020)","DOI":"10.1109\/CVPR42600.2020.00466"},{"key":"4071_CR19","doi-asserted-by":"publisher","first-page":"109913","DOI":"10.1016\/j.patcog.2023.109913","volume":"145","author":"J Shen","year":"2024","unstructured":"Shen, J., Chen, Y., Liu, Y., Zuo, X., Fan, H., Yang, W.: ICAFusion: Iterative cross-attention guided feature fusion for multispectral object detection. Pattern Recogn. 145, 109913 (2024)","journal-title":"Pattern Recogn."},{"key":"4071_CR20","doi-asserted-by":"crossref","unstructured":"Yu, J., Jiang, Y., Wang, Z., Cao, Z., Huang, T.: Unitbox: An advanced object detection network. In Proceedings of the 24th ACM international conference on multimedia (pp. 516-520). (2016, October)","DOI":"10.1145\/2964284.2967274"},{"key":"4071_CR21","doi-asserted-by":"crossref","unstructured":"Zheng, Z., Wang, P., Liu, W., Li, J., Ye, R., Ren, D.: Distance-IoU loss: Faster and better learning for bounding box regression. In Proceedings of the AAAI Conference on Artificial Intelligence (Vol. 34, No. 07, pp. 12993-13000). (2020, April)","DOI":"10.1609\/aaai.v34i07.6999"},{"key":"4071_CR22","doi-asserted-by":"publisher","first-page":"146","DOI":"10.1016\/j.neucom.2022.07.042","volume":"506","author":"YF Zhang","year":"2022","unstructured":"Zhang, Y.F., Ren, W., Zhang, Z., Jia, Z., Wang, L., Tan, T.: Focal and efficient IoU loss for accurate bounding box regression. Neurocomputing 506, 146\u2013157 (2022)","journal-title":"Neurocomputing"},{"key":"4071_CR23","doi-asserted-by":"crossref","unstructured":"Qin, X., Li, N., Weng, C., Su, D., Li, M.: Simple attention module based speaker verification with iterative noisy label detection. In ICASSP 2022-2022 IEEE international conference on acoustics, Speech and Signal Processing (ICASSP) (pp. 6722-6726). IEEE. (2022, May)","DOI":"10.1109\/ICASSP43922.2022.9746294"},{"key":"4071_CR24","doi-asserted-by":"crossref","unstructured":"Liu, J., Fan, X., Huang, Z., Wu, G., Liu, R., Zhong, W., Luo, Z.: Target-aware dual adversarial learning and a multi-scenario multi-modality benchmark to fuse infrared and visible for object detection. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 5802\u20135811). (2022)","DOI":"10.1109\/CVPR52688.2022.00571"},{"issue":"7","key":"4071_CR25","doi-asserted-by":"publisher","first-page":"4060","DOI":"10.1109\/LRA.2023.3272269","volume":"8","author":"M Liang","year":"2023","unstructured":"Liang, M., Hu, J., Bao, C., Feng, H., Deng, F., Lam, T.L.: Explicit attention-enhanced fusion for RGB-thermal perception tasks. IEEE Robot. Automat. Lett. 8(7), 4060\u20134067 (2023)","journal-title":"IEEE Robot. Automat. Lett."},{"key":"4071_CR26","doi-asserted-by":"crossref","unstructured":"Jia, X., Zhu, C., Li, M., Tang, W., Zhou, W.: LLVIP: A visible-infrared paired dataset for low-light vision. In IEEE\/cvf international conference on computer vision workshops, ICCVW 2021, Montreal, BC, Canada, October 11-17, 2021 (pp. 3489\u20133497). IEEE. (2021)","DOI":"10.1109\/ICCVW54120.2021.00389"},{"key":"4071_CR27","unstructured":"Zhao, T., Yuan, M., Wei, X.: Removal and selection: Improving RGB-infrared object detection via coarse-to-fine fusion. CoRR, arXiv: abs\/2401.10731. (2024)"},{"key":"4071_CR28","doi-asserted-by":"crossref","unstructured":"Wang, C.Y., Bochkovskiy, A., Liao, H.Y.M.: YOLOv7: Trainable bag-of-freebies sets new state-of-the-art for real-time object detectors. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 7464-7475). (2023)","DOI":"10.1109\/CVPR52729.2023.00721"},{"key":"4071_CR29","doi-asserted-by":"crossref","unstructured":"Zhao, Z., Xu, S., Zhang, C., Liu, J., Zhang, J., Li, P.: Didfuse: Deep image decomposition for infrared and visible image fusion. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 970\u2013976). (2020)","DOI":"10.24963\/ijcai.2020\/135"},{"issue":"10","key":"4071_CR30","doi-asserted-by":"publisher","first-page":"2761","DOI":"10.1007\/s11263-021-01501-8","volume":"129","author":"H Zhang","year":"2021","unstructured":"Zhang, H., Ma, J.: Sdnet: A versatile squeeze-and-decomposition network for real-time image fusion. Int. J. Comput. Vision 129(10), 2761\u20132785 (2021)","journal-title":"Int. J. Comput. Vision"},{"key":"4071_CR31","doi-asserted-by":"crossref","unstructured":"Xu, H., Ma, J., Yuan, J., Le, Z., Liu, W.: Rfnet: Unsupervised network for mutually reinforcing multimodal image registration and fusion. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR) (pp. 19647\u201319656). New Orleans, LA, USA, June 18-24, 2022. IEEE. (2022)","DOI":"10.1109\/CVPR52688.2022.01906"},{"key":"4071_CR32","doi-asserted-by":"crossref","unstructured":"Liu, J., Fan, X., Huang, Z., Wu, G., Liu, R., Zhong, W., Luo, Z.: Target-aware dual adversarial learning and a multi-scenario multi-modality benchmark to fuse infrared and visible for object detection. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 5802\u20135811). (2022)","DOI":"10.1109\/CVPR52688.2022.00571"},{"key":"4071_CR33","doi-asserted-by":"crossref","unstructured":"Sun, Y., Cao, B., Zhu, P., Hu, Q.: Detfusion: A detection-driven infrared and visible image fusion network. In MM \u201922: The 30th ACM international conference on multimedia, Lisboa, Portugal, October 10-14, 2022 (pp. 4003\u20134011). ACM. (2022)","DOI":"10.1145\/3503161.3547902"},{"key":"4071_CR34","doi-asserted-by":"crossref","unstructured":"Zhao, Z., Bai, H., Zhang, J., Zhang, Y., Xu, S., Lin, Z., Timofte, R., Van Gool, L.: Cddfuse: Correlation-driven dual-branch feature decomposition for multi-modality image fusion. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR) (pp. 5906\u20135916). Vancouver, BC, Canada, June 17-24, 2023. IEEE. (2023)","DOI":"10.1109\/CVPR52729.2023.00572"},{"key":"4071_CR35","doi-asserted-by":"crossref","unstructured":"Li, J., Chen, J., Liu, J., Ma, H.: Learning a graph neural network with cross modality interaction for image fusion. In Proceedings of the 31st ACM international conference on multimedia, MM 2023, Ottawa, ON, Canada, 29 October 2023 - 3 November 2023 (pp. 4471\u20134479). ACM. (2023)","DOI":"10.1145\/3581783.3612135"},{"issue":"12","key":"4071_CR36","doi-asserted-by":"publisher","first-page":"2121","DOI":"10.1109\/JAS.2022.106082","volume":"9","author":"L Tang","year":"2022","unstructured":"Tang, L., Deng, Y., Ma, Y., Huang, J., Ma, J.: Superfusion: a versatile image registration and fusion network with semantic awareness. IEEE\/CAA J. Automat. Sinica 9(12), 2121\u20132137 (2022)","journal-title":"IEEE\/CAA J. Automat. Sinica"},{"key":"4071_CR37","unstructured":"Ren, S., He, K., Girshick, R.B., Sun, J.: Faster R-CNN: Towards real-time object detection with region proposal networks. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 91\u201399). (2015)"},{"key":"4071_CR38","doi-asserted-by":"crossref","unstructured":"Cai, Z., Vasconcelos, N.: Cascade R-CNN: Delving into high quality object detection. In Proceedings of the 2018 IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (pp. 6154\u20136162). Salt Lake City, UT, USA, June 18-22, 2018. Computer vision foundation \/ IEEE Computer Society. (2018)","DOI":"10.1109\/CVPR.2018.00644"},{"key":"4071_CR39","doi-asserted-by":"crossref","unstructured":"Zhang, S., Wang, X., Wang, J., Pang, J., Lyu, C., Zhang, W., Luo, P., Chen, K.: Dense distinct query for end-to-end object detection. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR) 2023, Vancouver, BC, Canada, June 17-24, 2023 (pp. 7329\u20137338). IEEE. (2023)","DOI":"10.1109\/CVPR52729.2023.00708"},{"key":"4071_CR40","doi-asserted-by":"crossref","unstructured":"Liu, J., Zhang, S., Wang, S., Metaxas, D.N.: Multispectral deep neural networks for pedestrian detection. (2016)","DOI":"10.5244\/C.30.73"},{"key":"4071_CR41","doi-asserted-by":"crossref","unstructured":"Zhang, H., Fromont, \u00c9., Lef\u00e8vre, S., Avignon, B.: Guided attentive feature fusion for multispectral pedestrian detection. In Proceedings of the IEEE winter conference on applications of computer vision (WACV) 2021, Waikoloa, HI, USA, January 3-8, 2021 (pp. 72\u201380). IEEE. (2021)","DOI":"10.1109\/WACV48630.2021.00012"},{"key":"4071_CR42","doi-asserted-by":"crossref","unstructured":"Chen, Y.-T., Shi, J., Ye, Z., Mertz, C., Ramanan, D., Kong, S.: Multimodal object detection via probabilistic ensembling. In Proceedings of the 17th European conference on computer vision (ECCV) 2022, Tel Aviv, Israel, October 23-27, 2022, Part IX (pp. 139\u2013158). Springer. (2022)","DOI":"10.1007\/978-3-031-20077-9_9"},{"key":"4071_CR43","doi-asserted-by":"crossref","unstructured":"Cao, Y., Bin, J., Hamari, J., Blasch, E., Liu, Z.: Multimodal object detection by channel switching and spatial attention. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR) 2023 - Workshops, Vancouver, BC, Canada, June 17-24, 2023 (pp. 403\u2013411). IEEE. (2023)","DOI":"10.1109\/CVPRW59228.2023.00046"},{"key":"4071_CR44","doi-asserted-by":"publisher","first-page":"477","DOI":"10.1016\/j.inffus.2022.10.034","volume":"91","author":"L Tang","year":"2023","unstructured":"Tang, L., Xiang, X., Zhang, H., Gong, M., Ma, J.: Divfusion: Darkness-free infrared and visible image fusion. Info. Fusion 91, 477\u2013493 (2023)","journal-title":"Info. Fusion"}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-04071-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-025-04071-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-04071-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,24]],"date-time":"2025-09-24T14:01:32Z","timestamp":1758722492000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-025-04071-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,24]]},"references-count":44,"journal-issue":{"issue":"13","published-print":{"date-parts":[[2025,10]]}},"alternative-id":["4071"],"URL":"https:\/\/doi.org\/10.1007\/s00371-025-04071-9","relation":{"has-preprint":[{"id-type":"doi","id":"10.21203\/rs.3.rs-5279878\/v1","asserted-by":"object"}]},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"value":"0178-2789","type":"print"},{"value":"1432-2315","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,7,24]]},"assertion":[{"value":"17 June 2025","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 July 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}