{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,17]],"date-time":"2026-06-17T16:40:50Z","timestamp":1781714450963,"version":"3.54.5"},"reference-count":53,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2026,2,27]],"date-time":"2026-02-27T00:00:00Z","timestamp":1772150400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,2,27]],"date-time":"2026-02-27T00:00:00Z","timestamp":1772150400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100007129","name":"Natural Science Foundation of Shandong Province","doi-asserted-by":"publisher","award":["ZR2021MF060"],"award-info":[{"award-number":["ZR2021MF060"]}],"id":[{"id":"10.13039\/501100007129","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012165","name":"Key Technologies Research and Development Program","doi-asserted-by":"publisher","award":["2023YFC3321601"],"award-info":[{"award-number":["2023YFC3321601"]}],"id":[{"id":"10.13039\/501100012165","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61702303"],"award-info":[{"award-number":["61702303"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"20th Student Research Training Program (SRTP) at Shandong University","award":["2025507, A25093"],"award-info":[{"award-number":["2025507, A25093"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2026,3]]},"DOI":"10.1007\/s00371-026-04386-1","type":"journal-article","created":{"date-parts":[[2026,2,27]],"date-time":"2026-02-27T10:46:30Z","timestamp":1772189190000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Multi-scale cross-attention network for enhanced face forgery detection"],"prefix":"10.1007","volume":"42","author":[{"given":"Jin","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chengyou","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiao","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yupeng","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,2,27]]},"reference":[{"issue":"11","key":"4386_CR1","doi-asserted-by":"publisher","first-page":"139","DOI":"10.1145\/3422622","volume":"63","author":"I Goodfellow","year":"2020","unstructured":"Goodfellow, I., Pouget-Abadie, J., Mirza, M., Xu, B., Warde-Farley, D., Ozair, S., Courville, A., Bengio, Y.: Generative adversarial networks. Commun. ACM 63(11), 139\u2013144 (2020). https:\/\/doi.org\/10.1145\/3422622","journal-title":"Commun. ACM"},{"issue":"12","key":"4386_CR2","doi-asserted-by":"publisher","first-page":"4217","DOI":"10.1109\/TPAMI.2020.2970919","volume":"43","author":"T Karras","year":"2021","unstructured":"Karras, T., Laine, S., Aila, T.: A style-based generator architecture for generative adversarial networks. IEEE Trans. Pattern Anal. Mach. Intell. 43(12), 4217\u20134228 (2021). https:\/\/doi.org\/10.1109\/TPAMI.2020.2970919","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"4386_CR3","unstructured":"Ho, J., Jain, A., Abbeel, P.: Denoising diffusion probabilistic models. In: Annual Conference on Neural Information Processing Systems, 6\u201312 December (2020)"},{"key":"4386_CR4","doi-asserted-by":"publisher","first-page":"157","DOI":"10.1007\/s00371-024-03316-3","volume":"41","author":"C Li","year":"2025","unstructured":"Li, C., Da, F.: Refined dense face alignment through image matching. Vis. Comput. 41, 157\u2013171 (2025). https:\/\/doi.org\/10.1007\/s00371-024-03316-3","journal-title":"Vis. Comput."},{"key":"4386_CR5","doi-asserted-by":"crossref","unstructured":"Li, Y., Yang, X., Sun, P., Qi, H., Lyu, S.: Celeb-DF: A large-scale challenging dataset for deepfake forensics. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 4\u201319 June (2020)","DOI":"10.1109\/CVPR42600.2020.00327"},{"key":"4386_CR6","doi-asserted-by":"crossref","unstructured":"Hao, J., Zhang, Z., Yang, S., Xie, D., Pu, S.: TransForensics: Image forgery localization with dense self-attention. In: IEEE\/CVF International Conference on Computer Vision, 10\u201317 October (2021)","DOI":"10.1109\/ICCV48922.2021.01478"},{"key":"4386_CR7","doi-asserted-by":"crossref","unstructured":"Li, Y., Chang, M.-C., Lyu, S.: In ICTU Oculi: Exposing AI created fake videos by detecting eye blinking. In: IEEE International Workshop on Information Forensics and Security, 10\u201313 December (2018)","DOI":"10.1109\/WIFS.2018.8630787"},{"key":"4386_CR8","doi-asserted-by":"crossref","unstructured":"Yang, X., Li, Y., Lyu, S.: Exposing deep fakes using inconsistent head poses. In: IEEE International Conference on Acoustics, Speech and Signal Processing, 12\u201317 May (2019)","DOI":"10.1109\/ICASSP.2019.8683164"},{"key":"4386_CR9","doi-asserted-by":"crossref","unstructured":"Afchar, D., Nozick, V., Yamagishi, J., Echizen, I.: MesoNet: A compact facial video forgery detection network. In: IEEE International Workshop on Information Forensics and Security, 10\u201313 December (2018)","DOI":"10.1109\/WIFS.2018.8630761"},{"key":"4386_CR10","doi-asserted-by":"crossref","unstructured":"Rossler, A., Cozzolino, D., Verdoliva, L., Riess, C., Thies, J., Niessner, M.: FaceForensics++: Learning to detect manipulated facial images. In: IEEE\/CVF International Conference on Computer Vision, 27 October\u20132 November (2018)","DOI":"10.1109\/ICCV.2019.00009"},{"key":"4386_CR11","doi-asserted-by":"crossref","unstructured":"Qian, Y., Yin, G., Sheng, L., Chen, Z., Shao, J.: Thinking in frequency: Face forgery detection by mining frequency-aware clues. In: European Conference on Computer Vision, 23\u201328 August (2020)","DOI":"10.1007\/978-3-030-58610-2_6"},{"key":"4386_CR12","doi-asserted-by":"crossref","unstructured":"Liu, H., Li, X., Zhou, W., Chen, Y., He, Y., Xue, H., Zhang, W., Yu, N.: Spatial-phase shallow learning: Rethinking face forgery detection in frequency domain. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 20\u201325 June (2021)","DOI":"10.1109\/CVPR46437.2021.00083"},{"key":"4386_CR13","doi-asserted-by":"crossref","unstructured":"Zhao, H., Zhou, W., Chen, D., Wei, T., Zhang, W., Yu, N.: Multi-attentional deepfake detection. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 20\u201325 June (2021)","DOI":"10.1109\/CVPR46437.2021.00222"},{"key":"4386_CR14","doi-asserted-by":"crossref","unstructured":"Li, J., Xie, H., Li, J., Wang, Z., Zhang, Y.: Frequency-aware discriminative feature learning supervised by single-center loss for face forgery detection. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 19\u201325 June (2021)","DOI":"10.1109\/CVPR46437.2021.00639"},{"key":"4386_CR15","doi-asserted-by":"crossref","unstructured":"Luo, Y., Zhang, Y., Yan, J., Liu, W.: Generalizing face forgery detection with high-frequency features. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 19\u201325 June (2021)","DOI":"10.1109\/CVPR46437.2021.01605"},{"key":"4386_CR16","doi-asserted-by":"crossref","unstructured":"Wang, J., Wu, Z., Ouyang, W., Han, X., Chen, J., Jiang, Y.-G., Li, S.-N.: M2TR: Multi-modal multi-scale transformers for deepfake detection. In: International Conference on Multimedia Retrieval, 27\u201330 June (2022)","DOI":"10.1145\/3512527.3531415"},{"key":"4386_CR17","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2022.119361","volume":"215","author":"Z Guo","year":"2023","unstructured":"Guo, Z., Yang, G., Zhang, D., Xia, M.: Rethinking gradient operator for exposing AI-enabled face forgeries. Expert Syst. Appl. 215, 119361 (2023). https:\/\/doi.org\/10.1016\/j.eswa.2022.119361","journal-title":"Expert Syst. Appl."},{"key":"4386_CR18","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Goyal, P., Girshick, R., He, K., Doll\u00e1r, P.: Focal loss for dense object detection. In: IEEE\/CVF International Conference on Computer Vision, 22\u201329 October (2017)","DOI":"10.1109\/ICCV.2017.324"},{"issue":"4","key":"4386_CR19","doi-asserted-by":"publisher","first-page":"838","DOI":"10.1137\/0330046","volume":"30","author":"BT Polyak","year":"1992","unstructured":"Polyak, B.T., Juditsky, A.B.: Acceleration of stochastic approximation by averaging. SIAM J. Control. Optim. 30(4), 838\u2013855 (1992). https:\/\/doi.org\/10.1137\/0330046","journal-title":"SIAM J. Control. Optim."},{"key":"4386_CR20","doi-asserted-by":"crossref","unstructured":"Zi, B., Chang, M., Chen, J., Ma, X., Jiang, Y.-G.: WildDeepfake: A challenging real-world dataset for deepfake detection. In: 28th ACM International Conference on Multimedia, 12\u201316 October (2020)","DOI":"10.1145\/3394171.3413769"},{"key":"4386_CR21","doi-asserted-by":"publisher","unstructured":"Dolhansky, B., Howes, R., Pflaum, B., Baram, N., Ferrer, C.C.: The deepfake detection challenge (DFDC) preview dataset. arXiv preprint arXiv:1910.08854 (2019) https:\/\/doi.org\/10.48550\/arXiv.1910.08854","DOI":"10.48550\/arXiv.1910.08854"},{"key":"4386_CR22","doi-asserted-by":"crossref","unstructured":"Bonettini, N., Cannas, E., Mandelli, S., Bondi, L., Bestagini, P., Tubaro, S.: Video face manipulation detection through ensemble of CNNs. In: International Conference on Pattern Recognition, 10\u201315 January (2021)","DOI":"10.1109\/ICPR48806.2021.9412711"},{"key":"4386_CR23","doi-asserted-by":"crossref","unstructured":"Li, L., Bao, J., Zhang, T., Yang, H., Chen, D., Wen, F., Guo, B.: Face X-ray for more general face forgery detection. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 14\u201319 June (2020)","DOI":"10.1109\/CVPR42600.2020.00505"},{"key":"4386_CR24","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2023.110077","volume":"147","author":"C Zhu","year":"2024","unstructured":"Zhu, C., Zhang, B., Yin, Q., Yin, C., Lu, W.: Deepfake detection via inter-frame inconsistency recomposition and enhancement. Pattern Recognit. 147, 110077 (2024). https:\/\/doi.org\/10.1016\/j.patcog.2023.110077","journal-title":"Pattern Recognit."},{"key":"4386_CR25","doi-asserted-by":"publisher","first-page":"3405","DOI":"10.1109\/TMM.2023.3310341","volume":"26","author":"Y Yu","year":"2024","unstructured":"Yu, Y., Ni, R., Yang, S., Zhao, Y., Kot, A.C.: Narrowing domain gaps with bridging samples for generalized face forgery detection. IEEE Trans. Multimed. 26, 3405\u20133417 (2024). https:\/\/doi.org\/10.1109\/TMM.2023.3310341","journal-title":"IEEE Trans. Multimed."},{"issue":"2","key":"4386_CR26","doi-asserted-by":"publisher","first-page":"1255","DOI":"10.1109\/TCSVT.2023.3289147","volume":"34","author":"Z Guo","year":"2024","unstructured":"Guo, Z., Wang, L., Yang, W., Yang, G., Li, K.: LDFnet: lightweight dynamic fusion network for face forgery detection by integrating local artifacts and global texture information. IEEE Trans. Circuits Syst. Video Technol. 34(2), 1255\u20131265 (2024). https:\/\/doi.org\/10.1109\/TCSVT.2023.3289147","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"4386_CR27","doi-asserted-by":"crossref","unstructured":"Tan, C., Zhao, Y., Wei, S., Gu, G., Liu, P., Wei, Y.: Frequency-aware deepfake detection: Improving generalizability through frequency space learning. In: AAAI Conference on Artifical Intelligence, 20\u201327 February (2024)","DOI":"10.1609\/aaai.v38i5.28310"},{"key":"4386_CR28","doi-asserted-by":"crossref","unstructured":"Cao, J., Ma, C., Yao, T., Chen, S., Ding, S., Yang, X.: End-to-end reconstruction-classification learning for face forgery detection. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 18\u201324 June (2022)","DOI":"10.1109\/CVPR52688.2022.00408"},{"key":"4386_CR29","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2023.127192","volume":"572","author":"M Yu","year":"2024","unstructured":"Yu, M., Li, H., Yang, J., Li, X., Li, S., Zhang, J.: FDML: feature disentangling and multi-view learning for face forgery detection. Neurocomputing 572, 127192 (2024). https:\/\/doi.org\/10.1016\/j.neucom.2023.127192","journal-title":"Neurocomputing"},{"issue":"1","key":"4386_CR30","doi-asserted-by":"publisher","first-page":"11","DOI":"10.1007\/s40747-024-01634-6","volume":"11","author":"H Duan","year":"2025","unstructured":"Duan, H., Jiang, Q., Jin, X., Wozniak, M., Zhao, Y., Wu, L., Yao, S., Zhou, W.: Mf-net: multi-feature fusion network based on two-stream extraction and multi-scale enhancement for face forgery detection. Complex Intell Syst 11(1), 11 (2025). https:\/\/doi.org\/10.1007\/s40747-024-01634-6","journal-title":"Complex Intell Syst"},{"key":"4386_CR31","doi-asserted-by":"publisher","first-page":"7049","DOI":"10.1007\/s00371-024-03791-8","volume":"41","author":"M Uddin","year":"2025","unstructured":"Uddin, M., Fu, Z., Zhang, X.: Deepfake face detection via multi-level discrete wavelet transform and vision transformer. Vis. Comput. 41, 7049\u20137061 (2025). https:\/\/doi.org\/10.1007\/s00371-024-03791-8","journal-title":"Vis. Comput."},{"key":"4386_CR32","doi-asserted-by":"publisher","first-page":"3814","DOI":"10.1109\/TIFS.2024.3372773","volume":"19","author":"J Tian","year":"2024","unstructured":"Tian, J., Chen, P., Yu, C., Fu, X., Wang, X., Dai, J., Han, J.: Learning to discover forgery cues for face forgery detection. IEEE Trans. Inf. Forensics Secur. 19, 3814\u20133828 (2024). https:\/\/doi.org\/10.1109\/TIFS.2024.3372773","journal-title":"IEEE Trans. Inf. Forensics Secur."},{"issue":"10","key":"4386_CR33","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2024.123732","volume":"249","author":"J Gao","year":"2024","unstructured":"Gao, J., Xia, Z., Marcialis, G.L., Dang, C., Dai, J., Feng, X.: Deepfake detection based on high-frequency enhancement network for highly compressed content. Expert Syst. Appl. 249(10), 123732 (2024). https:\/\/doi.org\/10.1016\/j.eswa.2024.123732","journal-title":"Expert Syst. Appl."},{"key":"4386_CR34","doi-asserted-by":"publisher","DOI":"10.1016\/j.engappai.2025.112373","volume":"162","author":"C Shi","year":"2025","unstructured":"Shi, C., Wang, C., Zhou, X., Qin, Z.: Multi-modality complementary learning network with cross-modality interaction and adaptive fusion for face forgery detection. Eng. Appl. Artif. Intell. 162, 112373 (2025). https:\/\/doi.org\/10.1016\/j.engappai.2025.112373","journal-title":"Eng. Appl. Artif. Intell."},{"key":"4386_CR35","doi-asserted-by":"publisher","first-page":"401","DOI":"10.1109\/TIFS.2023.3324739","volume":"19","author":"Z Guo","year":"2024","unstructured":"Guo, Z., Jia, Z., Wang, L., Wang, D., Yang, G., Kasabov, N.: Constructing new backbone networks via space-frequency interactive convolution for deepfake detection. IEEE Trans. Inf. Forensic Secur. 19, 401\u2013413 (2024). https:\/\/doi.org\/10.1109\/TIFS.2023.3324739","journal-title":"IEEE Trans. Inf. Forensic Secur."},{"key":"4386_CR36","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2025.127108","volume":"276","author":"K Zhou","year":"2025","unstructured":"Zhou, K., Sun, G., Wang, J., Yu, L., Li, T.: MH-FFNet: leveraging mid-high frequency information for robust fine-grained face forgery detection. Expert Syst. Appl. 276, 127108 (2025). https:\/\/doi.org\/10.1016\/j.eswa.2025.127108","journal-title":"Expert Syst. Appl."},{"key":"4386_CR37","doi-asserted-by":"publisher","first-page":"1668","DOI":"10.1109\/TIP.2023.3246793","volume":"32","author":"Y Hua","year":"2023","unstructured":"Hua, Y., Shi, R., Wang, P., Ge, S.: Learning patch-channel correspondence for interpretable face forgery detection. IEEE Trans. Image Process. 32, 1668\u20131680 (2023). https:\/\/doi.org\/10.1109\/TIP.2023.3246793","journal-title":"IEEE Trans. Image Process."},{"key":"4386_CR38","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2025.114280","author":"Y Zhang","year":"2025","unstructured":"Zhang, Y., Wang, C., Zhou, X.: MSER-Net: multi-stage edge refinement network for deepfake detection. Knowl.-Based Syst. (2025). https:\/\/doi.org\/10.1016\/j.knosys.2025.114280","journal-title":"Knowl.-Based Syst."},{"issue":"9","key":"4386_CR39","doi-asserted-by":"publisher","first-page":"8972","DOI":"10.1109\/TCSVT.2024.3390945","volume":"34","author":"D Zhang","year":"2024","unstructured":"Zhang, D., Chen, J., Liao, X., Li, F., Chen, J., Yang, G.: Face forgery detection via multi-feature fusion and local enhancement. IEEE Trans. Circuits Syst. Video Technol. 34(9), 8972\u20138977 (2024). https:\/\/doi.org\/10.1109\/TCSVT.2024.3390945","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"4386_CR40","doi-asserted-by":"crossref","unstructured":"Cui, Y., Jia, M., Lin, T.-Y., Song, Y., Belongie, S.: Class-balanced loss based on effective number of samples. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 6\u201320 June (2019)","DOI":"10.1109\/CVPR.2019.00949"},{"key":"4386_CR41","doi-asserted-by":"publisher","unstructured":"Izmailov, P., Podoprikhin, D., Garipov, T., Vetrov, D.P., Wilson, A.G.: Averaging weights leads to wider optima and better generalization. arXiv preprint arXiv:1803.05407 (2018) https:\/\/doi.org\/10.48550\/arXiv.1803.05407","DOI":"10.48550\/arXiv.1803.05407"},{"issue":"6","key":"4386_CR42","doi-asserted-by":"publisher","first-page":"4257","DOI":"10.1109\/TCSVT.2023.3330390","volume":"34","author":"D Zhang","year":"2024","unstructured":"Zhang, D., Fu, C., Lu, D., Li, J., Zhang, Y.: Bi-source reconstruction-based classification network for face forgery video detection. IEEE Trans. Circuits Syst. Video Technol. 34(6), 4257\u20134269 (2024). https:\/\/doi.org\/10.1109\/TCSVT.2023.3330390","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"4386_CR43","doi-asserted-by":"publisher","unstructured":"Oquab, M., Darcet, T., Moutakanni, T., Vo, H., Szafraniec, M., Khalidov, V., Fernandez, P., Haziza, D., Massa, F., El-Nouby, A., Assran, M., Ballas, N., Galuba, W., Howes, R., Huang, P.-Y., Li, S.-W., Misra, I., Rabbat, M., Sharma, V., Synnaeve, G., Xu, H., J\u00e9gou, H., Mairal, J., Labatut, P., Joulin, A., Bojanowski, P.: DINOv2: Learning robust visual features without supervision. arXiv preprint arXiv:2304.07193 (2023) https:\/\/doi.org\/10.48550\/arXiv.2304.07193","DOI":"10.48550\/arXiv.2304.07193"},{"key":"4386_CR44","doi-asserted-by":"crossref","unstructured":"Ouyang, D., He, S., Zhang, G., Luo, M., Guo, H., Zhan, J., Huang, Z.: Efficient multi-scale attention module with cross-spatial learning. In: IEEE International Conference on Acoustics, Speech and Signal Processing, 4\u201310 June 2023 (2023)","DOI":"10.1109\/ICASSP49357.2023.10096516"},{"issue":"10","key":"4386_CR45","doi-asserted-by":"publisher","first-page":"1499","DOI":"10.1109\/LSP.2016.2603342","volume":"23","author":"K Zhang","year":"2016","unstructured":"Zhang, K., Zhang, Z., Li, Z., Qiao, Y.: Joint face detection and alignment using multi-task cascaded convolutional networks. IEEE Signal Process. Lett. 23(10), 1499\u20131503 (2016). https:\/\/doi.org\/10.1109\/LSP.2016.2603342","journal-title":"IEEE Signal Process. Lett."},{"key":"4386_CR46","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. In: International Conference on Learning Representations, 6\u20139 May (2019)"},{"key":"4386_CR47","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 26 June\u20131 July (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"4386_CR48","unstructured":"Tan, M., Le, Q.: EfficientNet: Rethinking model scaling for convolutional neural networks. In: International Conference on Machine Learning, 9\u201315 June (2019)"},{"key":"4386_CR49","doi-asserted-by":"publisher","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., Uszkoreit, J., Houlsby, N.: An image is worth $$16 \\times 16$$ words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2021) https:\/\/doi.org\/10.48550\/arXiv.2010.11929","DOI":"10.48550\/arXiv.2010.11929"},{"key":"4386_CR50","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., Guo, B.: Swin transformer: Hierarchical vision transformer using shifted windows. In: IEEE\/CVF International Conference on Computer Vision, 10\u201317 October (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"4386_CR51","volume-title":"Deep Learning","author":"I Goodfellow","year":"2016","unstructured":"Goodfellow, I., Bengio, Y., Courville, A.: Deep Learning. MIT Press, Cambridge, MA (2016)"},{"key":"4386_CR52","first-page":"2579","volume":"9","author":"L Maaten","year":"2008","unstructured":"Maaten, L., Hinton, G.: Visualizing data using t-SNE. J. Mach. Learn. Res. 9, 2579\u20132605 (2008)","journal-title":"J. Mach. Learn. Res."},{"key":"4386_CR53","doi-asserted-by":"crossref","unstructured":"Selvaraju, R.R., Cogswell, M., Das, A., Vedantam, R., Parikh, D., Batra, D.: Grad-CAM: Visual explanations from deep networks via gradient-based localization. In: IEEE\/CVF International Conference on Computer Vision, 22\u201329 October (2017)","DOI":"10.1109\/ICCV.2017.74"}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-026-04386-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-026-04386-1","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-026-04386-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,25]],"date-time":"2026-03-25T16:21:25Z","timestamp":1774455685000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-026-04386-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,2,27]]},"references-count":53,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2026,3]]}},"alternative-id":["4386"],"URL":"https:\/\/doi.org\/10.1007\/s00371-026-04386-1","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"value":"0178-2789","type":"print"},{"value":"1432-2315","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,2,27]]},"assertion":[{"value":"13 November 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 January 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 February 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no Conflict of interest to declare that are relevant to the content of this article.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"186"}}