{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,4]],"date-time":"2026-03-04T17:07:54Z","timestamp":1772644074323,"version":"3.50.1"},"reference-count":77,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,12,26]],"date-time":"2025-12-26T00:00:00Z","timestamp":1766707200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,12,26]],"date-time":"2025-12-26T00:00:00Z","timestamp":1766707200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62003065"],"award-info":[{"award-number":["62003065"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100005230","name":"Natural Science Foundation of Chongqing","doi-asserted-by":"crossref","award":["CSTB2024NSCQ-MSX0527"],"award-info":[{"award-number":["CSTB2024NSCQ-MSX0527"]}],"id":[{"id":"10.13039\/501100005230","id-type":"DOI","asserted-by":"crossref"}]},{"name":"Science and Technology Research Program of Chongqing Municipal Education Commission","award":["KJQN202200564"],"award-info":[{"award-number":["KJQN202200564"]}]},{"name":"Fund project of Chongqing Normal University","award":["21XLB032"],"award-info":[{"award-number":["21XLB032"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2026,1]]},"DOI":"10.1007\/s00371-025-04270-4","type":"journal-article","created":{"date-parts":[[2025,12,26]],"date-time":"2025-12-26T19:09:45Z","timestamp":1766776185000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["CSI-DMT: multi-focus image fusion via cross-task semantic interaction and dual-attention mixing transformer"],"prefix":"10.1007","volume":"42","author":[{"given":"Hao","family":"Zhai","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuanzhe","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhi","family":"Zeng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Minyu","family":"Deng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yiyang","family":"Ru","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,12,26]]},"reference":[{"key":"4270_CR1","doi-asserted-by":"publisher","first-page":"4816","DOI":"10.1109\/TIP.2020.2976190","volume":"29","author":"J Li","year":"2020","unstructured":"Li, J., Guo, X., Lu, G., Zhang, B., Xu, Y., Wu, F., Zhang, D.: DRPL: Deep regression pair learning for multi-focus image fusion. IEEE Trans. Image Process. 29, 4816\u20134831 (2020). https:\/\/doi.org\/10.1109\/TIP.2020.2976190","journal-title":"IEEE Trans. Image Process."},{"issue":"2","key":"4270_CR2","doi-asserted-by":"publisher","first-page":"220","DOI":"10.1109\/LSP.2014.2354534","volume":"22","author":"L Cao","year":"2014","unstructured":"Cao, L., Jin, L., Tao, H., Li, G., Zhuang, Z., Zhang, Y.: Multi-focus image fusion based on spatial frequency in discrete cosine transform domain. IEEE Signal Process. Lett. 22(2), 220\u2013224 (2014). https:\/\/doi.org\/10.1109\/LSP.2014.2354534","journal-title":"IEEE Signal Process. Lett."},{"key":"4270_CR3","doi-asserted-by":"publisher","first-page":"107325","DOI":"10.1016\/j.patcog.2020.107325","volume":"104","author":"Q Zhang","year":"2020","unstructured":"Zhang, Q., Li, G., Cao, Y., Han, J.: Multi-focus image fusion based on non-negative sparse representation and patch-level consistency rectification. Pattern Recognit. 104, 107325 (2020). https:\/\/doi.org\/10.1016\/j.patcog.2020.107325","journal-title":"Pattern Recognit."},{"key":"4270_CR4","doi-asserted-by":"publisher","first-page":"147","DOI":"10.1016\/j.inffus.2014.09.004","volume":"24","author":"Y Liu","year":"2015","unstructured":"Liu, Y., Liu, S., Wang, Z.: A general framework for image fusion based on multi-scale transform and sparse representation. Inf. Fusion 24, 147\u2013164 (2015). https:\/\/doi.org\/10.1016\/j.inffus.2014.09.004","journal-title":"Inf. Fusion"},{"key":"4270_CR5","doi-asserted-by":"publisher","first-page":"139","DOI":"10.1016\/j.inffus.2014.05.004","volume":"23","author":"Y Liu","year":"2015","unstructured":"Liu, Y., Liu, S., Wang, Z.: Multi-focus image fusion with dense SIFT. Inf. Fusion 23, 139\u2013155 (2015). https:\/\/doi.org\/10.1016\/j.inffus.2014.05.004","journal-title":"Inf. Fusion"},{"key":"4270_CR6","doi-asserted-by":"publisher","first-page":"271","DOI":"10.1007\/s11760-017-1155-y","volume":"12","author":"V Chaudhary","year":"2018","unstructured":"Chaudhary, V., Kumar, V.: Block-based image fusion using multi-scale analysis to enhance depth of field and dynamic range. Signal Image Video Process. 12, 271\u2013279 (2018). https:\/\/doi.org\/10.1007\/s11760-017-1155-y","journal-title":"Signal Image Video Process."},{"key":"4270_CR7","doi-asserted-by":"publisher","first-page":"119","DOI":"10.1016\/j.inffus.2018.07.010","volume":"48","author":"B Meher","year":"2019","unstructured":"Meher, B., Agrawal, S., Panda, R., Abraham, A.: A survey on region-based image fusion methods. Inf. Fusion 48, 119\u2013132 (2019). https:\/\/doi.org\/10.1016\/j.inffus.2018.07.010","journal-title":"Inf. Fusion"},{"key":"4270_CR8","doi-asserted-by":"publisher","first-page":"191","DOI":"10.1016\/j.inffus.2016.12.001","volume":"36","author":"Y Liu","year":"2017","unstructured":"Liu, Y., Chen, X., Peng, H., Wang, Z.: Multi-focus image fusion with a deep convolutional neural network. Inf. Fusion 36, 191\u2013207 (2017). https:\/\/doi.org\/10.1016\/j.inffus.2016.12.001","journal-title":"Inf. Fusion"},{"key":"4270_CR9","doi-asserted-by":"publisher","first-page":"125","DOI":"10.1016\/j.ins.2017.12.043","volume":"433","author":"H Tang","year":"2018","unstructured":"Tang, H., Xiao, B., Li, W., Wang, G.: Pixel convolutional neural network for multi-focus image fusion. Inf. Sci. 433, 125\u2013141 (2018). https:\/\/doi.org\/10.1016\/j.ins.2017.12.043","journal-title":"Inf. Sci."},{"key":"4270_CR10","doi-asserted-by":"publisher","first-page":"201","DOI":"10.1016\/j.inffus.2019.02.003","volume":"51","author":"M Amin-Naji","year":"2019","unstructured":"Amin-Naji, M., Aghagolzadeh, A., Ezoji, M.: Ensemble of cnn for multi-focus image fusion. Inf. Fusion 51, 201\u2013214 (2019). https:\/\/doi.org\/10.1016\/j.inffus.2019.02.003","journal-title":"Inf. Fusion"},{"issue":"8","key":"4270_CR11","doi-asserted-by":"publisher","first-page":"1982","DOI":"10.1109\/TMM.2019.2895292","volume":"21","author":"X Guo","year":"2019","unstructured":"Guo, X., Nie, R., Cao, J., Zhou, D., Mei, L., He, K.: FuseGAN: Learning to fuse multi-focus image via conditional generative adversarial network. IEEE Trans. Multimedia 21(8), 1982\u20131996 (2019). https:\/\/doi.org\/10.1109\/TMM.2019.2895292","journal-title":"IEEE Trans. Multimedia"},{"key":"4270_CR12","doi-asserted-by":"publisher","first-page":"127","DOI":"10.1016\/j.inffus.2022.11.014","volume":"92","author":"X Hu","year":"2023","unstructured":"Hu, X., Jiang, J., Liu, X., Ma, J.: ZMFF: Zero-shot multi-focus image fusion. Inf. Fusion 92, 127\u2013138 (2023). https:\/\/doi.org\/10.1016\/j.inffus.2022.11.014","journal-title":"Inf. Fusion"},{"key":"4270_CR13","doi-asserted-by":"publisher","first-page":"99","DOI":"10.1016\/j.inffus.2019.07.011","volume":"54","author":"Y Zhang","year":"2020","unstructured":"Zhang, Y., Liu, Y., Sun, P., Yan, H., Zhao, X., Zhang, L.: IFCNN: A general image fusion framework based on convolutional neural network. Inf. Fusion 54, 99\u2013118 (2020). https:\/\/doi.org\/10.1016\/j.inffus.2019.07.011","journal-title":"Inf. Fusion"},{"key":"4270_CR14","doi-asserted-by":"publisher","first-page":"40","DOI":"10.1016\/j.inffus.2020.08.022","volume":"66","author":"H Zhang","year":"2021","unstructured":"Zhang, H., Le, Z., Shao, Z., Xu, H., Ma, J.: MFF-GAN: an unsupervised generative adversarial network with adaptive and gradient joint constraints for multi-focus image fusion. Inf. Fusion 66, 40\u201353 (2021). https:\/\/doi.org\/10.1016\/j.inffus.2020.08.022","journal-title":"Inf. Fusion"},{"key":"4270_CR15","doi-asserted-by":"publisher","first-page":"121664","DOI":"10.1016\/j.eswa.2023.121664","volume":"238","author":"M Li","year":"2024","unstructured":"Li, M., Pei, R., Zheng, T., Zhang, Y., Fu, W.: FusionDiff: multi-focus image fusion using denoising diffusion probabilistic models. Expert Syst. Appl. 238, 121664 (2024). https:\/\/doi.org\/10.1016\/j.eswa.2023.121664","journal-title":"Expert Syst. Appl."},{"key":"4270_CR16","doi-asserted-by":"publisher","unstructured":"Ramachandran, P., Parmar, N., Vaswani, A., Bello, I., Levskaya, A., Shlens. J: Stand-alone self-attention in vision models, Advances in Neural Information Processing Systems 32 (2019) https:\/\/doi.org\/10.48550\/arXiv.1906.05909","DOI":"10.48550\/arXiv.1906.05909"},{"key":"4270_CR17","doi-asserted-by":"publisher","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers, in: European Conference on Computer Vision, Springer, pp. 213\u2013229, (2020). https:\/\/doi.org\/10.1007\/978-3-030-58452-8_13","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"4270_CR18","doi-asserted-by":"publisher","unstructured":"Zheng, S., Lu, J., Zhao, H., Zhu, X., Luo, Z., Wang, Y., Fu, Y., Feng, J., Xiang, T., Torr, P. H. et al.,: Rethinking semantic segmentation from a sequence-to-sequence perspective with transformers, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition,pp. 6881\u20136890, (2021). https:\/\/doi.org\/10.48550\/arXiv.2012.15840","DOI":"10.48550\/arXiv.2012.15840"},{"key":"4270_CR19","doi-asserted-by":"publisher","unstructured":"Zhao, H., Jiang, L., Jia, J., Torr, PH., Koltun, V.: Point transformer, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 16259\u201316268, (2021). https:\/\/doi.org\/10.48550\/arXiv.2012.09164","DOI":"10.48550\/arXiv.2012.09164"},{"key":"4270_CR20","doi-asserted-by":"publisher","unstructured":"Van Tulder, G., Tong,Y., Marchiori,E.: Multi-view analysis of unregistered medical images using cross-view transformers, in: Medical Image Computing and Computer Assisted Intervention\u2013MICCAI 2021: 24th International Conference, Strasbourg, France, September 27\u2013October1, 2021, Proceedings, Part III 24, Springer, 2021, pp. 104\u2013113, (2021). https:\/\/doi.org\/10.48550\/arXiv.2103.11390","DOI":"10.48550\/arXiv.2103.11390"},{"issue":"7","key":"4270_CR21","doi-asserted-by":"publisher","first-page":"1200","DOI":"10.1109\/JAS.2022.105686","volume":"9","author":"J Ma","year":"2022","unstructured":"Ma, J., Tang, L., Fan, F., Huang, J., Mei, X., Ma, Y.: SwinFusion: cross-domain long-range learning for general image fusion via Swin transformer. IEEE\/CAA J. Autom. Sin. 9(7), 1200\u20131217 (2022). https:\/\/doi.org\/10.1109\/JAS.2022.105686","journal-title":"IEEE\/CAA J. Autom. Sin."},{"key":"4270_CR22","doi-asserted-by":"publisher","first-page":"121156","DOI":"10.1016\/j.eswa.2023.121156","volume":"235","author":"Z Duan","year":"2024","unstructured":"Duan, Z., Luo, X., Zhang, T.: Combining transformers with CNN for multi-focus imagefusion. Expert Syst. Appl. 235, 121156 (2024). https:\/\/doi.org\/10.1016\/j.eswa.2023.121156","journal-title":"Expert Syst. Appl."},{"key":"4270_CR23","doi-asserted-by":"publisher","first-page":"102353","DOI":"10.1016\/j.displa.2022.102353","volume":"76","author":"P Wu","year":"2023","unstructured":"Wu, P., Jiang, L., Hua, Z., Li, J.: Multi-focus image fusion: transformer and shallow feature attention matters. Displays 76, 102353 (2023). https:\/\/doi.org\/10.1016\/j.displa.2022.102353","journal-title":"Displays"},{"key":"4270_CR24","doi-asserted-by":"publisher","first-page":"107688","DOI":"10.1016\/j.cmpb.2023.107688","volume":"240","author":"R Pei","year":"2023","unstructured":"Pei, R., Yao, K., Xu, X., Zhang, X., Yang, X., Fu, W., Zhang, Y.: TransFusion-net for multifocus microscopic biomedical image fusion. Comput. Methods Programs Biomed. 240, 107688 (2023). https:\/\/doi.org\/10.1016\/j.cmpb.2023.107688","journal-title":"Comput. Methods Programs Biomed."},{"issue":"2","key":"4270_CR25","doi-asserted-by":"publisher","first-page":"102","DOI":"10.2478\/msr-2014-0014","volume":"14","author":"Y Yang","year":"2014","unstructured":"Yang, Y., Huang, S., Gao, J., Qian, Z.: Multi-focus image fusion using an effective discrete wavelet transform based algorithm. Meas. Sci. Rev. 14(2), 102 (2014). https:\/\/doi.org\/10.2478\/msr-2014-0014","journal-title":"Meas. Sci. Rev."},{"key":"4270_CR26","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.neucom.2015.10.084","volume":"182","author":"B Yu","year":"2016","unstructured":"Yu, B., Jia, B., Ding, L., Cai, Z., Wu, Q., Law, R., Huang, J., Song, L., Fu, S.: Hybrid dual-tree complex wavelet transform and support vector machine for digital multi-focus image fusion. Neurocomputing 182, 1\u20139 (2016). https:\/\/doi.org\/10.1016\/j.neucom.2015.10.084","journal-title":"Neurocomputing"},{"key":"4270_CR27","doi-asserted-by":"publisher","first-page":"299","DOI":"10.1016\/j.patcog.2018.06.003","volume":"83","author":"Q Zhang","year":"2018","unstructured":"Zhang, Q., Shi, T., Wang, F., Blum, R.S., Han, J.: Robust sparse representation based multi-focus image fusion with dictionary construction and local spatial consistency. Pattern Recogn. 83, 299\u2013313 (2018). https:\/\/doi.org\/10.1016\/j.patcog.2018.06.003","journal-title":"Pattern Recogn."},{"issue":"7","key":"4270_CR28","doi-asserted-by":"publisher","first-page":"522","DOI":"10.3390\/e20070522","volume":"20","author":"Y Li","year":"2018","unstructured":"Li, Y., Sun, Y., Huang, X., Qi, G., Zheng, M., Zhu, Z.: An image fusion method based on sparse representation and sum modified-Laplacian in NSCT domain. Entropy 20(7), 522 (2018). https:\/\/doi.org\/10.3390\/e20070522","journal-title":"Entropy"},{"issue":"12","key":"4270_CR29","doi-asserted-by":"publisher","first-page":"8861","DOI":"10.1016\/j.eswa.2010.06.011","volume":"37","author":"V Aslantas","year":"2010","unstructured":"Aslantas, V., Kurban, R.: Fusion of multi-focus images using differential evolution algorithm. Expert Syst. Appl. 37(12), 8861\u20138870 (2010). https:\/\/doi.org\/10.1016\/j.eswa.2010.06.011","journal-title":"Expert Syst. Appl."},{"issue":"2","key":"4270_CR30","doi-asserted-by":"publisher","first-page":"136","DOI":"10.1016\/j.inffus.2012.01.007","volume":"14","author":"I De","year":"2013","unstructured":"De, I., Chanda, B.: Multi-focus image fusion using a morphology-based focus measure in a quad-tree structure. Inf. Fusion 14(2), 136\u2013146 (2013). https:\/\/doi.org\/10.1016\/j.inffus.2012.01.007","journal-title":"Inf. Fusion"},{"key":"4270_CR31","doi-asserted-by":"publisher","first-page":"108782","DOI":"10.1016\/j.aml.2023.108782","volume":"145","author":"Y-W Wen","year":"2023","unstructured":"Wen, Y.-W., Xiao, X., Chen, Y.: A weighted gradient model with total variation regularization for multi-focus image fusion. Appl. Math. Lett. 145, 108782 (2023). https:\/\/doi.org\/10.1016\/j.aml.2023.108782","journal-title":"Appl. Math. Lett."},{"key":"4270_CR32","doi-asserted-by":"publisher","first-page":"116295","DOI":"10.1016\/j.image.2021.116295","volume":"96","author":"Y Wang","year":"2021","unstructured":"Wang, Y., Xu, S., Liu, J., Zhao, Z., Zhang, C., Zhang, J.: MFIF-GAN: a new generative adversarial network for multi-focus image fusion. Signal Process. Image Commun. 96, 116295 (2021). https:\/\/doi.org\/10.1016\/j.image.2021.116295","journal-title":"Signal Process. Image Commun."},{"key":"4270_CR33","doi-asserted-by":"publisher","first-page":"309","DOI":"10.1109\/TCI.2021.3063872","volume":"7","author":"J Ma","year":"2021","unstructured":"Ma, J., Le, Z., Tian, X., Jiang, J.: SMFuse: multi-focus image fusion via self-supervised mask-optimization. IEEE Trans. Comput. Imaging 7, 309\u2013320 (2021). https:\/\/doi.org\/10.1109\/TCI.2021.3063872","journal-title":"IEEE Trans. Comput. Imaging"},{"key":"4270_CR34","doi-asserted-by":"publisher","first-page":"5793","DOI":"10.1007\/s00521-020-05358-9","volume":"33","author":"B Ma","year":"2021","unstructured":"Ma, B., Zhu, Y., Yin, X., Ban, X., Huang, H., Mukeshimana, M.: Sesf-fuse: an unsupervised deep model for multi-focus image fusion. Neural Comput. Appl. 33, 5793\u20135804 (2021). https:\/\/doi.org\/10.1007\/s00521-020-05358-9","journal-title":"Neural Comput. Appl."},{"issue":"1","key":"4270_CR35","doi-asserted-by":"publisher","first-page":"502","DOI":"10.1109\/TPAMI.2020.3012548","volume":"44","author":"H Xu","year":"2020","unstructured":"Xu, H., Ma, J., Jiang, J., Guo, X., Ling, H.: U2Fusion: a unified unsupervised image fusion network. IEEE Trans. Pattern Anal. Mach. Intell. 44(1), 502\u2013518 (2020). https:\/\/doi.org\/10.1109\/TPAMI.2020.3012548","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"9","key":"4270_CR36","doi-asserted-by":"publisher","first-page":"101751","DOI":"10.1016\/j.jksuci.2023.101751","volume":"35","author":"S Liu","year":"2023","unstructured":"Liu, S., Peng, W., Liu, Y., Zhao, J., Su, Y., Zhang, Y.: Afcanet: an adaptive feature concatenate attention network for multi-focus image fusion. J. King Saud Univ. Comput. Inf. Sci. 35(9), 101751 (2023). https:\/\/doi.org\/10.1016\/j.jksuci.2023.101751","journal-title":"J. King Saud Univ. Comput. Inf. Sci."},{"key":"4270_CR37","doi-asserted-by":"publisher","first-page":"123244","DOI":"10.1016\/j.eswa.2024.123244","volume":"247","author":"Y Qi","year":"2024","unstructured":"Qi, Y., Yang, Z., Lu, X., Li, S., Ma, Y.: A multi-channel neural network model for multi-focus image fusion. Expert Syst. Appl. 247, 123244 (2024). https:\/\/doi.org\/10.1016\/j.eswa.2024.123244","journal-title":"Expert Syst. Appl."},{"key":"4270_CR38","doi-asserted-by":"publisher","first-page":"106603","DOI":"10.1016\/j.neunet.2024.106603","volume":"179","author":"B Li","year":"2024","unstructured":"Li, B., Zhang, L., Liu, J., Peng, H., Wang, Q., Liu, J.: Multi-focus image fusion with parameter adaptive dual channel dynamic threshold neural P systems. Neural Netw. 179, 106603 (2024). https:\/\/doi.org\/10.1016\/j.neunet.2024.106603","journal-title":"Neural Netw."},{"key":"4270_CR39","doi-asserted-by":"publisher","first-page":"121772","DOI":"10.1016\/j.ins.2024.121772","volume":"698","author":"L Fang","year":"2025","unstructured":"Fang, L., Hou, M., Huang, B., Chen, G., Yang, J.: Dcafusion: a novel general image fusion framework based on reference image reconstruction and dual-cross attention mechanism. Inf. Sci. 698, 121772 (2025). https:\/\/doi.org\/10.1016\/j.ins.2024.121772","journal-title":"Inf. Sci."},{"key":"4270_CR40","doi-asserted-by":"publisher","first-page":"111041","DOI":"10.1016\/j.patcog.2024.111041","volume":"158","author":"X Wang","year":"2025","unstructured":"Wang, X., Fang, L., Zhao, J., Pan, Z., Li, H., Li, Y.: Mmae: a universal image fusion method via mask attention mechanism. Pattern Recogn. 158, 111041 (2025). https:\/\/doi.org\/10.1016\/j.patcog.2024.111041","journal-title":"Pattern Recogn."},{"key":"4270_CR41","doi-asserted-by":"publisher","first-page":"121389","DOI":"10.1016\/j.eswa.2023.121389","volume":"237","author":"C Wang","year":"2024","unstructured":"Wang, C., Zang, Y., Zhou, D., Mei, J., Nie, R., Zhou, L.: Robust multi-focus image fusion using focus property detection and deep image matting. Expert Syst. Appl. 237, 121389 (2024). https:\/\/doi.org\/10.1016\/j.eswa.2023.121389","journal-title":"Expert Syst. Appl."},{"key":"4270_CR42","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2025.102974","author":"K Zheng","year":"2025","unstructured":"Zheng, K., Cheng, J., Liu, Y.: Unfolding coupled convolutional sparse representation formulti-focus image fusion. Inf. Fusion (2025). https:\/\/doi.org\/10.1016\/j.inffus.2025.102974","journal-title":"Inf. Fusion"},{"key":"4270_CR43","doi-asserted-by":"publisher","first-page":"102125","DOI":"10.1016\/j.inffus.2023.102125","volume":"103","author":"P Chen","year":"2024","unstructured":"Chen, P., Jiang, J., Li, L., Yao, J.: A defocus and similarity attention-based cascaded network for multi-focus and misaligned image fusion. Inf. Fusion 103, 102125 (2024). https:\/\/doi.org\/10.1016\/j.inffus.2023.102125","journal-title":"Inf. Fusion"},{"key":"4270_CR44","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2022.3227717","author":"Y Liu","year":"2023","unstructured":"Liu, Y., Zhang, Y., Wang, Y., Hou, F., Yuan, J., Tian, J., Zhang, Y., Shi, Z., Fan, J., He, Z.: A survey of visual transformers. IEEE Trans Neural Networks Learning Syst (2023). https:\/\/doi.org\/10.1109\/TNNLS.2022.3227717","journal-title":"IEEE Trans Neural Networks Learning Syst"},{"key":"4270_CR45","doi-asserted-by":"publisher","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, G., Heigold, S., Gelly, et al., An image is worth 16\u00d716 words: Transformers for image recognition at scale, arXiv preprint arXiv:2010.11929 (2020), https:\/\/doi.org\/10.48550\/arXiv.2010.11929","DOI":"10.48550\/arXiv.2010.11929"},{"key":"4270_CR46","doi-asserted-by":"publisher","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., Guo,B.,: Swin transformer:Hierarchical vision transformer using shifted windows, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 10012\u201310022, (2021). https:\/\/doi.org\/10.48550\/arXiv.2103.14030","DOI":"10.48550\/arXiv.2103.14030"},{"key":"4270_CR47","doi-asserted-by":"publisher","unstructured":"He, K., Chen, X., Xie, S., Li, Y., Doll\u00e1r, P,. Girshick, R.: Masked autoencoders are scalable vision learners, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 16000\u201316009, (2022). https:\/\/doi.org\/10.48550\/arXiv.2111.06377","DOI":"10.48550\/arXiv.2111.06377"},{"key":"4270_CR48","doi-asserted-by":"publisher","unstructured":"L. Yuan, Y. Chen, T. Wang, W. Yu, Y. Shi, Z.-H. Jiang, F. E. H. Tay, J. Feng, S. Yan.: Tokens-to-token ViT: Training vision transformers from scratch on ImageNet, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 558\u2013567 (2021). https:\/\/doi.org\/10.48550\/arXiv.2101.11986","DOI":"10.48550\/arXiv.2101.11986"},{"key":"4270_CR49","doi-asserted-by":"publisher","unstructured":"W. Wang, E. Xie, X. Li, D.-P. Fan, K. Song, D. Liang, T. Lu, P. Luo, L. Shao.: Pyramid vision transformer: A versatile backbone for dense prediction without convolutions, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 568\u2013578, (2021). https:\/\/doi.org\/10.48550\/arXiv.2102.12122","DOI":"10.48550\/arXiv.2102.12122"},{"key":"4270_CR50","doi-asserted-by":"publisher","unstructured":"H. Wu, B. Xiao, N. Codella, M. Liu, X. Dai, L. Yuan, L. Zhang.: CVT: Introducing convolutions to vision transformers, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 22\u201331 (2021). https:\/\/doi.org\/10.48550\/arXiv.2103.15808","DOI":"10.48550\/arXiv.2103.15808"},{"key":"4270_CR51","doi-asserted-by":"publisher","first-page":"102837","DOI":"10.1016\/j.displa.2024.102837","volume":"85","author":"H Zhai","year":"2024","unstructured":"Zhai, H., Ouyang, Y., Luo, N., Chen, L., Zeng, Z.: Msi-dtrans: a multi-focus image fusion using multilayer semantic interaction and dynamic transformer. Displays 85, 102837 (2024). https:\/\/doi.org\/10.1016\/j.displa.2024.102837","journal-title":"Displays"},{"key":"4270_CR52","doi-asserted-by":"publisher","first-page":"107967","DOI":"10.1016\/j.engappai.2024.107967","volume":"133","author":"H Zhai","year":"2024","unstructured":"Zhai, H., Zheng, W., Ouyang, Y., Pan, X., Zhang, W.: Multi-focus image fusion via interactive transformer and asymmetric soft sharing. Eng. Appl. Artif. Intell. 133, 107967 (2024). https:\/\/doi.org\/10.1016\/j.engappai.2024.107967","journal-title":"Eng. Appl. Artif. Intell."},{"key":"4270_CR53","doi-asserted-by":"publisher","unstructured":"Ioffe, S., Szegedy, C.: Batch normalization: accelerating deep network training by reducing internal covariate shift, in: International Conference on Machine Learning, 2015, pp. 448\u2013456, (2015). https:\/\/doi.org\/10.48550\/arXiv.1502.03167","DOI":"10.48550\/arXiv.1502.03167"},{"key":"4270_CR54","doi-asserted-by":"publisher","unstructured":"Yu, F., Koltun, V.: Multi-scale context aggregation by dilated convolutions, arXiv preprint arXiv:1511.07122 (2015). https:\/\/doi.org\/10.48550\/arXiv.1511.07122","DOI":"10.48550\/arXiv.1511.07122"},{"key":"4270_CR55","doi-asserted-by":"publisher","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2016, pp. 770\u2013778, (2016). https:\/\/doi.org\/10.1109\/CVPR.2016.90","DOI":"10.1109\/CVPR.2016.90"},{"key":"4270_CR56","doi-asserted-by":"publisher","unstructured":"Z. Liu, H. Hu, Y. Lin, Z. Yao, Z. Xie, Y. Wei, J. Ning, Y. Cao, Z. Zhang, L. Dong, et al.,: Swin transformer v2: Scaling up capacity and resolution, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 12009\u201312019, (2022). https:\/\/doi.org\/10.48550\/arXiv.2111.09883","DOI":"10.48550\/arXiv.2111.09883"},{"key":"4270_CR57","doi-asserted-by":"publisher","unstructured":"M. Sandler, A. Howard, M. Zhu, A. Zhmoginov, L.-C. Chen.: MobilenetV2: Inverted residuals and linear bottlenecks, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2018, pp. 4510\u20134520, (2018). https:\/\/doi.org\/10.1109\/CVPR.2018.00474","DOI":"10.1109\/CVPR.2018.00474"},{"key":"4270_CR58","doi-asserted-by":"publisher","unstructured":"F. Chollet, Xception: Deep learning with depthwise separable convolutions, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2017, pp. 1251\u20131258, (2017). https:\/\/doi.org\/10.1109\/CVPR.2017.195","DOI":"10.1109\/CVPR.2017.195"},{"key":"4270_CR59","doi-asserted-by":"publisher","unstructured":"J. L. Ba, J. R. Kiros, G. E. Hinton.: Layer normalization, arXiv preprint arXiv:1607.06450 (2016). https:\/\/doi.org\/10.48550\/arXiv.1607.06450","DOI":"10.48550\/arXiv.1607.06450"},{"key":"4270_CR60","doi-asserted-by":"publisher","unstructured":"S. Woo, J. Park, J.-Y. Lee, I. S. Kweon : CBAM: Convolutional block attention module, in: Proceedings of the European Conference on Computer Vision (ECCV), (2018), pp. 3\u201319, https:\/\/doi.org\/10.1007\/978-3-030-01234-2_1","DOI":"10.1007\/978-3-030-01234-2_1"},{"key":"4270_CR61","doi-asserted-by":"publisher","unstructured":"R. Hou, H. Chang, B. Ma, S. Shan, X. Chen,: Cross attention network for few-shot classification, Advances in Neural Information Processing Systems 32 (2019). https:\/\/doi.org\/10.48550\/arXiv.1910.07677","DOI":"10.48550\/arXiv.1910.07677"},{"issue":"1","key":"4270_CR62","doi-asserted-by":"publisher","first-page":"47","DOI":"10.1109\/TCI.2016.2644865","volume":"3","author":"H Zhao","year":"2016","unstructured":"Zhao, H., Gallo, O., Frosio, I., Kautz, J.: Loss functions for image restoration with neural networks. IEEE Trans. Comput. Imaging 3(1), 47\u201357 (2016). https:\/\/doi.org\/10.1109\/TCI.2016.2644865","journal-title":"IEEE Trans. Comput. Imaging"},{"key":"4270_CR63","doi-asserted-by":"publisher","unstructured":"Wang, L., Lu, H., Wang, Y., Feng, M., Wang, D., Yin, B. Ruan, X.: Learning to detect salient objects with image-level supervision, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2017, pp. 136\u2013145, (2017). https:\/\/doi.org\/10.1109\/CVPR.2017.404","DOI":"10.1109\/CVPR.2017.404"},{"key":"4270_CR64","doi-asserted-by":"publisher","first-page":"72","DOI":"10.1016\/j.inffus.2014.10.004","volume":"25","author":"M Nejati","year":"2015","unstructured":"Nejati, M., Samavi, S., Shirani, S.: Multi-focus image fusion using dictionary-based sparse representation. Inf. Fusion 25, 72\u201384 (2015). https:\/\/doi.org\/10.1016\/j.inffus.2014.10.004","journal-title":"Inf. Fusion"},{"key":"4270_CR65","doi-asserted-by":"publisher","unstructured":"Xu, S., Wei, X., Zhang, C., Liu, J., Zhang, J.: MFFW: A new dataset for multi-focus image fusion, arXiv preprint arXiv:2002.04780 (2020). https:\/\/doi.org\/10.48550\/arXiv.2002.04780","DOI":"10.48550\/arXiv.2002.04780"},{"key":"4270_CR66","doi-asserted-by":"publisher","first-page":"370","DOI":"10.1016\/j.patrec.2020.08.002","volume":"138","author":"J Zhang","year":"2020","unstructured":"Zhang, J., Liao, Q., Liu, S., Ma, H., Yang, W., Xue, J.-H.: Real-MFF: a large realistic multi-focus image dataset with ground truth. Pattern Recognit. Lett. 138, 370\u2013377 (2020). https:\/\/doi.org\/10.1016\/j.patrec.2020.08.002","journal-title":"Pattern Recognit. Lett."},{"key":"4270_CR67","doi-asserted-by":"publisher","unstructured":"Loshchilov, I., Hutter, I.: Decoupled weight decay regularization, arXiv preprint arXiv:1711.05101 (2017). https:\/\/doi.org\/10.48550\/arXiv.1711.05101","DOI":"10.48550\/arXiv.1711.05101"},{"issue":"12","key":"4270_CR68","doi-asserted-by":"publisher","first-page":"2959","DOI":"10.1109\/26.477498","volume":"43","author":"AM Eskicioglu","year":"2002","unstructured":"Eskicioglu, A.M., Fisher, P.S.: Image quality measures and their performance. IEEE Trans. Commun. 43(12), 2959\u20132965 (2002). https:\/\/doi.org\/10.1109\/26.477498","journal-title":"IEEE Trans. Commun."},{"issue":"7","key":"4270_CR69","doi-asserted-by":"publisher","first-page":"313","DOI":"10.1049\/el:20020212","volume":"38","author":"G Qu","year":"2002","unstructured":"Qu, G., Zhang, D., Yan, P.: Information measure for performance of image fusion. Electron. Lett. 38(7), 313\u2013315 (2002). https:\/\/doi.org\/10.1049\/el:20020212","journal-title":"Electron. Lett."},{"key":"4270_CR70","doi-asserted-by":"publisher","unstructured":"Petrovic, V., Xydeas,C.: Objective image fusion performance characterisation, In: Proceedings of the Tenth IEEE International Conference on Computer Vision (ICCV\u201905) Volume 1, 2005, pp. 1866\u20131871, (2005). https:\/\/doi.org\/10.1109\/ICCV.2005.175","DOI":"10.1109\/ICCV.2005.175"},{"issue":"12","key":"4270_CR71","doi-asserted-by":"publisher","first-page":"1890","DOI":"10.1016\/j.aeue.2015.09.004","volume":"69","author":"V Aslantas","year":"2015","unstructured":"Aslantas, V., Bendes, E.: A new image quality metric for image fusion: the sum of the correlations of differences. AEU-Int. J. Electron. Commun. 69(12), 1890\u20131896 (2015). https:\/\/doi.org\/10.1016\/j.aeue.2015.09.004","journal-title":"AEU-Int. J. Electron. Commun."},{"issue":"4","key":"4270_CR72","doi-asserted-by":"publisher","first-page":"600","DOI":"10.1109\/TIP.2003.819861","volume":"13","author":"Z Wang","year":"2004","unstructured":"Wang, Z., Bovik, A.C., Sheikh, H.R., Simoncelli, E.P.: Image quality assessment: from error visibility to structural similarity. IEEE Trans. Image Process. 13(4), 600\u2013612 (2004). https:\/\/doi.org\/10.1109\/TIP.2003.819861","journal-title":"IEEE Trans. Image Process."},{"issue":"2","key":"4270_CR73","doi-asserted-by":"publisher","first-page":"127","DOI":"10.1016\/j.inffus.2011.08.002","volume":"14","author":"Y Han","year":"2013","unstructured":"Han, Y., Cai, Y., Cao, Y., Xu, X.: A new image fusion performance metric based on visual information fidelity. Inf. Fusion 14(2), 127\u2013135 (2013). https:\/\/doi.org\/10.1016\/j.inffus.2011.08.002","journal-title":"Inf. Fusion"},{"key":"4270_CR74","doi-asserted-by":"publisher","first-page":"125665","DOI":"10.1016\/j.eswa.2024.125665","volume":"262","author":"Y Ouyang","year":"2025","unstructured":"Ouyang, Y., Zhai, H., Hu, H., Li, X., Zeng, Z.: FusionGCN: Multi-focus image fusion using superpixel features generation GCN and pixel-level feature reconstruction CNN. Expert Syst. Appl. 262, 125665 (2025). https:\/\/doi.org\/10.1016\/j.eswa.2024.125665","journal-title":"Expert Syst. Appl."},{"key":"4270_CR75","doi-asserted-by":"publisher","first-page":"81","DOI":"10.1016\/j.inffus.2016.09.006","volume":"35","author":"Y Zhang","year":"2017","unstructured":"Zhang, Y., Bai, X., Wang, T.: Boundary finding based multi-focus image fusion throughmulti-scale morphological focus-measure. Inf. Fusion 35, 81\u2013101 (2017). https:\/\/doi.org\/10.1016\/j.inffus.2016.09.006","journal-title":"Inf. Fusion"},{"key":"4270_CR76","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1109\/TIM.2021.3124058","volume":"70","author":"Y Liu","year":"2021","unstructured":"Liu, Y., Wang, L., Cheng, J., Chen, X.: Multiscale feature interactive network for multifocus image fusion. IEEE Trans. Instrum. Meas. 70, 1\u201316 (2021). https:\/\/doi.org\/10.1109\/TIM.2021.3124058","journal-title":"IEEE Trans. Instrum. Meas."},{"key":"4270_CR77","doi-asserted-by":"publisher","first-page":"80","DOI":"10.1016\/j.inffus.2022.11.010","volume":"92","author":"C Cheng","year":"2023","unstructured":"Cheng, C., Xu, T., Wu, X.: MUFusion: a general unsupervised image fusion network based on memory unit. Inf. Fusion 92, 80\u201392 (2023). https:\/\/doi.org\/10.1016\/j.inffus.2022.11.010","journal-title":"Inf. Fusion"}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-04270-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-025-04270-4","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-04270-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,4]],"date-time":"2026-03-04T13:05:27Z","timestamp":1772629527000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-025-04270-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,26]]},"references-count":77,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2026,1]]}},"alternative-id":["4270"],"URL":"https:\/\/doi.org\/10.1007\/s00371-025-04270-4","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"value":"0178-2789","type":"print"},{"value":"1432-2315","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,12,26]]},"assertion":[{"value":"26 May 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"1 October 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 December 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"97"}}