{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,20]],"date-time":"2026-05-20T20:14:53Z","timestamp":1779308093310,"version":"3.51.4"},"reference-count":80,"publisher":"Springer Science and Business Media LLC","issue":"11","license":[{"start":{"date-parts":[[2025,5,8]],"date-time":"2025-05-08T00:00:00Z","timestamp":1746662400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,5,8]],"date-time":"2025-05-08T00:00:00Z","timestamp":1746662400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2025,9]]},"DOI":"10.1007\/s00371-025-03905-w","type":"journal-article","created":{"date-parts":[[2025,5,8]],"date-time":"2025-05-08T14:36:13Z","timestamp":1746714973000},"page":"8895-8910","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Semantically Enhanced Dual Visual Fusion Transformer for accurate image captioning"],"prefix":"10.1007","volume":"41","author":[{"given":"Juan","family":"Yang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hao","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ronggui","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lixia","family":"Xue","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiaping","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,5,8]]},"reference":[{"key":"3905_CR1","unstructured":"Sutskever, I., Vinyals, O., Le, Q.V.: Sequence to sequence learning with neural networks. In: Advances in Neural Information Processing Systems, vol. 27 (2014)"},{"key":"3905_CR2","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2023.119773","volume":"221","author":"D Sharma","year":"2023","unstructured":"Sharma, D., Dhiman, C., Kumar, D.: Evolution of visual data captioning methods, datasets, and evaluation metrics: a comprehensive survey. Expert Syst. Appl. 221, 119773 (2023)","journal-title":"Expert Syst. Appl."},{"key":"3905_CR3","doi-asserted-by":"crossref","unstructured":"Vinyals, O., Toshev, A., Bengio, S., Erhan, D.: Show and tell: a neural image caption generator. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3156\u20133164 (2015)","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"3905_CR4","doi-asserted-by":"crossref","unstructured":"Anderson, P., He, X., Buehler, C., Teney, D., Johnson, M., Gould, S., Zhang, L.: Bottom-up and top-down attention for image captioning and visual question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6077\u20136086 (2018)","DOI":"10.1109\/CVPR.2018.00636"},{"issue":"8","key":"3905_CR5","doi-asserted-by":"publisher","first-page":"1798","DOI":"10.1109\/TPAMI.2013.50","volume":"35","author":"Y Bengio","year":"2013","unstructured":"Bengio, Y., Courville, A., Vincent, P.: Representation learning: a review and new perspectives. IEEE Trans. Pattern Anal. Mach. Intell. 35(8), 1798\u20131828 (2013)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"3905_CR6","first-page":"1045","volume-title":"Interspeech","author":"T Mikolov","year":"2010","unstructured":"Mikolov, T., Karafi\u00e1t, M., Burget, L., Cernock\u1ef3, J., Khudanpur, S.: Recurrent neural network based language model. In: Interspeech, vol. 2, pp. 1045\u20131048. Makuhari, Chiba City (2010)"},{"key":"3905_CR7","unstructured":"Xu, K., Ba, J., Kiros, R., Cho, K., Courville, A., Salakhudinov, R., Zemel, R., Bengio, Y.: Show, attend and tell: neural image caption generation with visual attention. In: International Conference on Machine Learning, pp. 2048\u20132057. PMLR (2015)"},{"key":"3905_CR8","doi-asserted-by":"crossref","unstructured":"You, Q., Jin, H., Wang, Z., Fang, C., Luo, J.: Image captioning with semantic attention. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4651\u20134659 (2016)","DOI":"10.1109\/CVPR.2016.503"},{"key":"3905_CR9","doi-asserted-by":"crossref","unstructured":"Lu, J., Xiong, C., Parikh, D., Socher, R.: Knowing when to look: adaptive attention via a visual sentinel for image captioning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 375\u2013383 (2017)","DOI":"10.1109\/CVPR.2017.345"},{"key":"3905_CR10","unstructured":"Szegedy, C., Toshev, A., Erhan, D.: Deep neural networks for object detection. In: Advances in Neural Information Processing Systems, vol. 26 (2013)"},{"key":"3905_CR11","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster R-CNN: towards real-time object detection with region proposal networks. In: Advances in Neural Information Processing Systems, vol. 28 (2015)"},{"issue":"4","key":"3905_CR12","doi-asserted-by":"publisher","first-page":"277","DOI":"10.1007\/s10489-024-06073-x","volume":"55","author":"B Wang","year":"2025","unstructured":"Wang, B., Yang, M., Cao, P., Liu, Y.: A novel embedded cross framework for high-resolution salient object detection. Appl. Intell. 55(4), 277 (2025)","journal-title":"Appl. Intell."},{"key":"3905_CR13","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers. In: European Conference on Computer Vision, pp. 213\u2013229. Springer (2020)","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"3905_CR14","doi-asserted-by":"crossref","unstructured":"Hu, H., Gu, J., Zhang, Z., Dai, J., Wei, Y.: Relation networks for object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3588\u20133597 (2018)","DOI":"10.1109\/CVPR.2018.00378"},{"key":"3905_CR15","doi-asserted-by":"crossref","unstructured":"Luo, Y., Ji, J., Sun, X., Cao, L., Wu, Y., Huang, F., Lin, C.-W., Ji, R.: Dual-level collaborative transformer for image captioning. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 35, pp. 2286\u20132293 (2021)","DOI":"10.1609\/aaai.v35i3.16328"},{"key":"3905_CR16","doi-asserted-by":"crossref","unstructured":"Zhang, X., Sun, X., Luo, Y., Ji, J., Zhou, Y., Wu, Y., Huang, F., Ji, R.: Rstnet: captioning with adaptive attention on visual and non-visual words. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15465\u201315474 (2021)","DOI":"10.1109\/CVPR46437.2021.01521"},{"key":"3905_CR17","doi-asserted-by":"publisher","DOI":"10.32604\/cmc.2023.037861","author":"Z Tang","year":"2023","unstructured":"Tang, Z., Yi, Y., Yu, C., Yin, A.: Pcatnet: position-class awareness transformer for image captioning. Comput. Mater. Contin. (2023). https:\/\/doi.org\/10.32604\/cmc.2023.037861","journal-title":"Comput. Mater. Contin."},{"key":"3905_CR18","doi-asserted-by":"crossref","unstructured":"Guo, L., Liu, J., Zhu, X., Yao, P., Lu, S., Lu, H.: Normalized and geometry-aware self-attention network for image captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10327\u201310336 (2020)","DOI":"10.1109\/CVPR42600.2020.01034"},{"issue":"11","key":"3905_CR19","doi-asserted-by":"publisher","first-page":"7706","DOI":"10.1109\/TCSVT.2022.3181490","volume":"32","author":"W Jiang","year":"2022","unstructured":"Jiang, W., Zhou, W., Hu, H.: Double-stream position learning transformer network for image captioning. IEEE Trans. Circuits Syst. Video Technol. 32(11), 7706\u20137718 (2022)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"issue":"6","key":"3905_CR20","first-page":"509","volume":"5","author":"M Wang","year":"2023","unstructured":"Wang, M., Meng, M., Liu, J., Wu, J.: Learning adequate alignment and interaction for cross-modal retrieval. Virt. Real. Intell. Hardw. 5(6), 509\u2013522 (2023)","journal-title":"Virt. Real. Intell. Hardw."},{"key":"3905_CR21","doi-asserted-by":"publisher","DOI":"10.1007\/s00371-024-03729-0","author":"F Liang","year":"2024","unstructured":"Liang, F., Huang, Z., Wang, W., He, Z., En, Q.: Dynamic text prompt joint multimodal features for accurate plant disease image captioning. Vis. Comput. (2024). https:\/\/doi.org\/10.1007\/s00371-024-03729-0","journal-title":"Vis. Comput."},{"key":"3905_CR22","doi-asserted-by":"crossref","unstructured":"Sharma, D., Dhiman, C., Kumar, D.: Automated image caption generation framework using adaptive attention and bi-LSTM. In: 2022 IEEE Delhi Section Conference (DELCON), pp. 1\u20135. IEEE (2022)","DOI":"10.1109\/DELCON54057.2022.9752859"},{"key":"3905_CR23","unstructured":"Sharma, D., Dhiman, C., Kumar, D.: Unma-capsumt: unified and multi-head attention-driven caption summarization transformer. arXiv preprint arXiv:2412.11836 (2024)"},{"issue":"12","key":"3905_CR24","doi-asserted-by":"publisher","first-page":"18413","DOI":"10.1007\/s11042-021-10578-9","volume":"80","author":"C Sur","year":"2021","unstructured":"Sur, C.: MRRC: multiple role representation crossover interpretation for image captioning with R-CNN feature distribution composition (FDC). Multimed. Tools Appl. 80(12), 18413\u201318443 (2021)","journal-title":"Multimed. Tools Appl."},{"key":"3905_CR25","doi-asserted-by":"publisher","DOI":"10.1016\/j.physd.2019.132306","volume":"404","author":"A Sherstinsky","year":"2020","unstructured":"Sherstinsky, A.: Fundamentals of recurrent neural network (RNN) and long short-term memory (lSTM) network. Phys. D Nonlinear Phenom. 404, 132306 (2020)","journal-title":"Phys. D Nonlinear Phenom."},{"issue":"8","key":"3905_CR26","doi-asserted-by":"publisher","first-page":"5384","DOI":"10.1109\/TGRS.2019.2899129","volume":"57","author":"R Hang","year":"2019","unstructured":"Hang, R., Liu, Q., Hong, D., Ghamisi, P.: Cascaded recurrent neural networks for hyperspectral image classification. IEEE Trans. Geosci. Remote Sens. 57(8), 5384\u20135394 (2019)","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"issue":"6","key":"3905_CR27","doi-asserted-by":"publisher","DOI":"10.1007\/s11704-021-0236-9","volume":"16","author":"T Wang","year":"2022","unstructured":"Wang, T., Li, J., Wu, H.-N., Li, C., Snoussi, H., Wu, Y.: ReslNet: deep residual lSTM network with longer input for action recognition. Front. Comput. Sci. 16(6), 166334 (2022)","journal-title":"Front. Comput. Sci."},{"key":"3905_CR28","doi-asserted-by":"crossref","unstructured":"Basak, D., Srijith, P., Desarkar, M.S.: Transformer based multitask learning for image captioning and object detection. In: Pacific-Asia Conference on Knowledge Discovery and Data Mining, pp. 260\u2013272. Springer (2024)","DOI":"10.1007\/978-981-97-2253-2_21"},{"issue":"6","key":"3905_CR29","doi-asserted-by":"publisher","first-page":"1796","DOI":"10.3390\/s24061796","volume":"24","author":"D Abdal Hafeth","year":"2024","unstructured":"Abdal Hafeth, D., Kollias, S.: Insights into object semantics: leveraging transformer networks for advanced image captioning. Sensors 24(6), 1796 (2024)","journal-title":"Sensors"},{"issue":"1","key":"3905_CR30","doi-asserted-by":"publisher","first-page":"57","DOI":"10.1016\/j.vrih.2022.07.006","volume":"5","author":"M Zhang","year":"2023","unstructured":"Zhang, M., Tian, X.: Transformer architecture based on mutual attention for image-anomaly detection. Virt. Reality Intell. Hardw. 5(1), 57\u201367 (2023)","journal-title":"Virt. Reality Intell. Hardw."},{"issue":"1","key":"3905_CR31","doi-asserted-by":"publisher","first-page":"2201","DOI":"10.1002\/cav.2201","volume":"35","author":"X Zhu","year":"2024","unstructured":"Zhu, X., Yao, X., Zhang, J., Zhu, M., You, L., Yang, X., Zhang, J., Zhao, H., Zeng, D.: TMSDNet: transformer with multi-scale dense network for single and multi-view 3d reconstruction. Comput. Anim. Virt. Worlds 35(1), 2201 (2024)","journal-title":"Comput. Anim. Virt. Worlds"},{"key":"3905_CR32","doi-asserted-by":"publisher","first-page":"50","DOI":"10.1109\/TMM.2021.3120873","volume":"25","author":"X Lin","year":"2021","unstructured":"Lin, X., Sun, S., Huang, W., Sheng, B., Li, P., Feng, D.D.: EAPT: efficient attention pyramid transformer for image processing. IEEE Trans. Multimed. 25, 50\u201361 (2021)","journal-title":"IEEE Trans. Multimed."},{"issue":"5","key":"3905_CR33","doi-asserted-by":"publisher","first-page":"2293","DOI":"10.1002\/cav.2293","volume":"35","author":"Z Xiao","year":"2024","unstructured":"Xiao, Z., Chen, Y., Zhou, X., He, M., Liu, L., Yu, F., Jiang, M.: Human action recognition in immersive virtual reality based on multi-scale spatio-temporal attention network. Comput. Anim. Virt. Worlds 35(5), 2293 (2024)","journal-title":"Comput. Anim. Virt. Worlds"},{"issue":"1","key":"3905_CR34","doi-asserted-by":"publisher","first-page":"595","DOI":"10.1109\/TII.2019.2934144","volume":"16","author":"J Xu","year":"2019","unstructured":"Xu, J., Park, S.H., Zhang, X.: A temporally irreversible visual attention model inspired by motion sensitive neurons. IEEE Trans. Ind. Inform. 16(1), 595\u2013605 (2019)","journal-title":"IEEE Trans. Ind. Inform."},{"key":"3905_CR35","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-024-19315-4","author":"Y Ren","year":"2024","unstructured":"Ren, Y., Zhang, J., Xu, W., Lin, Y., Fu, B., Thanh, D.N.: Dual visual align-cross attention-based image captioning transformer. Multimed. Tools Appl. (2024). https:\/\/doi.org\/10.1007\/s11042-024-19315-4","journal-title":"Multimed. Tools Appl."},{"issue":"6","key":"3905_CR36","doi-asserted-by":"publisher","first-page":"2888","DOI":"10.1109\/TVCG.2023.3261935","volume":"29","author":"Y Li","year":"2023","unstructured":"Li, Y., Wang, J., Dai, X., Wang, L., Yeh, C.-C.M., Zheng, Y., Zhang, W., Ma, K.-L.: How does attention work in vision transformers? A visual analytics attempt. IEEE Trans. Vis. Comput. Graph. 29(6), 2888\u20132900 (2023)","journal-title":"IEEE Trans. Vis. Comput. Graph."},{"key":"3905_CR37","doi-asserted-by":"publisher","first-page":"4057","DOI":"10.1109\/TIP.2019.2956143","volume":"29","author":"K Wang","year":"2020","unstructured":"Wang, K., Peng, X., Yang, J., Meng, D., Qiao, Y.: Region attention networks for pose and occlusion robust facial expression recognition. IEEE Trans. Image Process. 29, 4057\u20134069 (2020)","journal-title":"IEEE Trans. Image Process."},{"key":"3905_CR38","doi-asserted-by":"publisher","DOI":"10.1016\/j.engappai.2021.104574","volume":"108","author":"Y Wei","year":"2022","unstructured":"Wei, Y., Wu, C., Li, G., Shi, H.: Sequential transformer via an outside-in attention for image captioning. Eng. Appl. Artif. Intell. 108, 104574 (2022)","journal-title":"Eng. Appl. Artif. Intell."},{"key":"3905_CR39","doi-asserted-by":"publisher","DOI":"10.1109\/LGRS.2024.3383163","author":"W Peng","year":"2024","unstructured":"Peng, W., Jian, P., Mao, Z., Zhao, Y.: Change captioning for satellite images time series. IEEE Geosci. Remote Sens. Lett. (2024). https:\/\/doi.org\/10.1109\/LGRS.2024.3383163","journal-title":"IEEE Geosci. Remote Sens. Lett."},{"key":"3905_CR40","unstructured":"Herdade, S., Kappeler, A., Boakye, K., Soares, J.: Image captioning: transforming objects into words. In: Advances in Neural Information Processing Systems, vol. 32 (2019)"},{"key":"3905_CR41","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Liu, C., Hao, F., Liu, Z.: Style transfer-based unsupervised change detection from heterogeneous images. In: IEEE Transactions on Aerospace and Electronic Systems (2025)","DOI":"10.1109\/TAES.2025.3529431"},{"issue":"1","key":"3905_CR42","doi-asserted-by":"publisher","first-page":"641","DOI":"10.1109\/TPAMI.2022.3148470","volume":"45","author":"K Li","year":"2022","unstructured":"Li, K., Zhang, Y., Li, K., Li, Y., Fu, Y.: Image-text embedding learning via visual and textual semantic reasoning. IEEE Trans. Pattern Anal. Mach. Intell. 45(1), 641\u2013656 (2022)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"6","key":"3905_CR43","doi-asserted-by":"publisher","first-page":"4794","DOI":"10.1007\/s10489-024-05416-y","volume":"54","author":"Q Su","year":"2024","unstructured":"Su, Q., Hu, J., Li, Z.: Visual contextual relationship augmented transformer for image captioning. Appl. Intell. 54(6), 4794\u20134813 (2024)","journal-title":"Appl. Intell."},{"key":"3905_CR44","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2024.110555","volume":"153","author":"S Cao","year":"2024","unstructured":"Cao, S., An, G., Cen, Y., Yang, Z., Lin, W.: Cast: cross-modal retrieval and visual conditioning for image captioning. Pattern Recognit. 153, 110555 (2024)","journal-title":"Pattern Recognit."},{"issue":"6","key":"3905_CR45","doi-asserted-by":"publisher","first-page":"1169","DOI":"10.1007\/s41095-023-0389-6","volume":"10","author":"S Deng","year":"2024","unstructured":"Deng, S., Wu, L., Shi, G., Xing, L., Jian, M., Xiang, Y., Dong, R.: Learning to compose diversified prompts for image emotion classification. Comput. Vis. Media 10(6), 1169\u20131183 (2024)","journal-title":"Comput. Vis. Media"},{"key":"3905_CR46","doi-asserted-by":"crossref","unstructured":"Chen, C.-F.R., Fan, Q., Panda, R.: Crossvit: cross-attention multi-scale vision transformer for image classification. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 357\u2013366 (2021)","DOI":"10.1109\/ICCV48922.2021.00041"},{"key":"3905_CR47","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, \u0141., Polosukhin, I.: Attention is all you need. In: Advances in nEural Information Processing Systems, vol. 30 (2017)"},{"key":"3905_CR48","unstructured":"Ke, G., He, D., Liu, T.-Y.: Rethinking positional encoding in language pre-training. arXiv preprint arXiv:2006.15595 (2020)"},{"key":"3905_CR49","doi-asserted-by":"publisher","DOI":"10.1016\/j.engappai.2023.106384","volume":"123","author":"J Hu","year":"2023","unstructured":"Hu, J., Yang, Y., An, Y., Yao, L.: Dual-spatial normalized transformer for image captioning. Eng. Appl. Artif. Intell. 123, 106384 (2023)","journal-title":"Eng. Appl. Artif. Intell."},{"key":"3905_CR50","doi-asserted-by":"publisher","DOI":"10.1016\/j.imavis.2022.104575","volume":"128","author":"J Hu","year":"2022","unstructured":"Hu, J., Yang, Y., Yao, L., An, Y., Pan, L.: Position-guided transformer for image captioning. Image Vis. Comput. 128, 104575 (2022)","journal-title":"Image Vis. Comput."},{"key":"3905_CR51","doi-asserted-by":"crossref","unstructured":"Shaw, P., Uszkoreit, J., Vaswani, A.: Self-attention with relative position representations. arXiv preprint arXiv:1803.02155 (2018)","DOI":"10.18653\/v1\/N18-2074"},{"key":"3905_CR52","doi-asserted-by":"crossref","unstructured":"Rennie, S.J., Marcheret, E., Mroueh, Y., Ross, J., Goel, V.: Self-critical sequence training for image captioning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7008\u20137024 (2017)","DOI":"10.1109\/CVPR.2017.131"},{"key":"3905_CR53","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r, P., Zitnick, C.L.: Microsoft coco: common objects in context. In: Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6\u201312, 2014, Proceedings, Part V 13, pp. 740\u2013755. Springer (2014)","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"3905_CR54","doi-asserted-by":"crossref","unstructured":"Karpathy, A., Fei-Fei, L.: Deep visual-semantic alignments for generating image descriptions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3128\u20133137 (2015)","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"3905_CR55","doi-asserted-by":"crossref","unstructured":"Papineni, K., Roukos, S., Ward, T., Zhu, W.-J.: Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics, pp. 311\u2013318 (2002)","DOI":"10.3115\/1073083.1073135"},{"key":"3905_CR56","unstructured":"Banerjee, S., Lavie, A.: Meteor: An automatic metric for MT evaluation with improved correlation with human judgments. In: Proceedings of the ACL Workshop on Intrinsic and Extrinsic Evaluation Measures for Machine Translation And\/or Summarization, pp. 65\u201372 (2005)"},{"key":"3905_CR57","unstructured":"Lin, C.-Y.: Rouge: a package for automatic evaluation of summaries. In: Text Summarization Branches Out, pp. 74\u201381 (2004)"},{"key":"3905_CR58","doi-asserted-by":"crossref","unstructured":"Vedantam, R., Lawrence\u00a0Zitnick, C., Parikh, D.: Cider: consensus-based image description evaluation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4566\u20134575 (2015)","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"3905_CR59","doi-asserted-by":"crossref","unstructured":"Jiang, H., Misra, I., Rohrbach, M., Learned-Miller, E., Chen, X.: In defense of grid features for visual question answering. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10267\u201310276 (2020)","DOI":"10.1109\/CVPR42600.2020.01028"},{"key":"3905_CR60","unstructured":"Kingma, D.P., Ba, J.: Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)"},{"key":"3905_CR61","doi-asserted-by":"crossref","unstructured":"Yao, T., Pan, Y., Li, Y., Mei, T.: Exploring visual relationship for image captioning. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 684\u2013699 (2018)","DOI":"10.1007\/978-3-030-01264-9_42"},{"key":"3905_CR62","doi-asserted-by":"crossref","unstructured":"Huang, L., Wang, W., Chen, J., Wei, X.-Y.: Attention on attention for image captioning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4634\u20134643 (2019)","DOI":"10.1109\/ICCV.2019.00473"},{"key":"3905_CR63","doi-asserted-by":"crossref","unstructured":"Cornia, M., Stefanini, M., Baraldi, L., Cucchiara, R.: Meshed-memory transformer for image captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10578\u201310587 (2020)","DOI":"10.1109\/CVPR42600.2020.01059"},{"key":"3905_CR64","doi-asserted-by":"crossref","unstructured":"Ji, J., Luo, Y., Sun, X., Chen, F., Luo, G., Wu, Y., Gao, Y., Ji, R.: Improving image captioning by leveraging intra-and inter-layer global representation in transformer network. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 35, pp. 1655\u20131663 (2021)","DOI":"10.1609\/aaai.v35i2.16258"},{"key":"3905_CR65","doi-asserted-by":"crossref","unstructured":"Pan, Y., Yao, T., Li, Y., Mei, T.: X-linear attention networks for image captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10971\u201310980 (2020)","DOI":"10.1109\/CVPR42600.2020.01098"},{"key":"3905_CR66","doi-asserted-by":"crossref","unstructured":"Fan, Z., Wei, Z., Wang, S., Wang, R., Li, Z., Shan, H., Huang, X.: Tcic: Theme concepts learning cross language and vision for image captioning. arXiv preprint arXiv:2106.10936 (2021)","DOI":"10.24963\/ijcai.2021\/91"},{"key":"3905_CR67","doi-asserted-by":"crossref","unstructured":"Gao, Y., Wang, N., Suo, W., Sun, M., Wang, P.: Improving image captioning via enhancing dual-side context awareness. In: Proceedings of the 2022 International Conference on Multimedia Retrieval, pp. 389\u2013397 (2022)","DOI":"10.1145\/3512527.3531379"},{"key":"3905_CR68","doi-asserted-by":"crossref","unstructured":"Jiang, Z., Wang, X., Zhai, Z., Cheng, B.: LG-MLFormer: local and global MlLP for image captioning. Int. J. Multimed. Inf. Retr. 12(1), 4 (2023)","DOI":"10.1007\/s13735-023-00266-9"},{"key":"3905_CR69","doi-asserted-by":"crossref","unstructured":"Chen, L., Yang, Y., Hu, J., Pan, L., Zhai, H.: Relational-convergent transformer for image captioning. Displays 77, 102377 (2023)","DOI":"10.1016\/j.displa.2023.102377"},{"key":"3905_CR70","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2024.3425513","author":"L Wang","year":"2024","unstructured":"Wang, L., Chen, H., Liu, Y., Lyu, Y.: Regular constrained multimodal fusion for image captioning. IEEE Trans. Circuits Syst. Video Technol. (2024). https:\/\/doi.org\/10.1109\/TCSVT.2024.3425513","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"3905_CR71","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2022.117174","volume":"201","author":"C Wang","year":"2022","unstructured":"Wang, C., Shen, Y., Ji, L.: Geometry attention transformer with position-aware lSTMs for image captioning. Expert Syst. Appl. 201, 117174 (2022)","journal-title":"Expert Syst. Appl."},{"issue":"2","key":"3905_CR72","doi-asserted-by":"publisher","first-page":"4219","DOI":"10.1007\/s11042-023-15291-3","volume":"83","author":"D Sharma","year":"2024","unstructured":"Sharma, D., Dhiman, C., Kumar, D.: XGL-T transformer model for intelligent image captioning. Multimed. Tools Appl. 83(2), 4219\u20134240 (2024)","journal-title":"Multimed. Tools Appl."},{"key":"3905_CR73","unstructured":"Xu, K., Ba, J., Kiros, R., Cho, K., Courville, A.C., Salakhutdinov, R., Zemel, R.S., Bengio, Y.: Show, attend and tell: Neural image caption generation with visual attention. In: International Conference on Machine Learning. https:\/\/api.semanticscholar.org\/CorpusID:1055111 (2015)"},{"key":"3905_CR74","doi-asserted-by":"crossref","unstructured":"Cheng, Y., Huang, F., Zhou, L., Jin, C., Zhang, Y., Zhang, T.: A hierarchical multimodal attention-based neural network for image captioning. In: Proceedings of the 40th International ACM SIGIR Conference on Research and Development in Information Retrieval (2017)","DOI":"10.1145\/3077136.3080671"},{"key":"3905_CR75","doi-asserted-by":"publisher","first-page":"726","DOI":"10.1109\/TMM.2017.2751140","volume":"20","author":"L Li","year":"2018","unstructured":"Li, L., Tang, S., Zhang, Y., Deng, L., Tian, Q.: GLA: global-local attention for image description. IEEE Trans. Multimed. 20, 726\u2013737 (2018)","journal-title":"IEEE Trans. Multimed."},{"key":"3905_CR76","doi-asserted-by":"publisher","first-page":"2942","DOI":"10.1109\/TMM.2019.2915033","volume":"21","author":"X Xiao","year":"2019","unstructured":"Xiao, X., Wang, L., Ding, K., Xiang, S., Pan, C.: Deep hierarchical encoder-decoder network for image captioning. IEEE Trans. Multimed. 21, 2942\u20132956 (2019)","journal-title":"IEEE Trans. Multimed."},{"key":"3905_CR77","doi-asserted-by":"publisher","first-page":"520","DOI":"10.1016\/j.neucom.2019.04.095","volume":"398","author":"S Ding","year":"2020","unstructured":"Ding, S., Qu, S., Xi, Y., Wan, S.: Stimulus-driven and concept-driven analysis for image caption generation. Neurocomputing 398, 520\u2013530 (2020)","journal-title":"Neurocomputing"},{"key":"3905_CR78","doi-asserted-by":"publisher","first-page":"6575","DOI":"10.1007\/s10489-021-02734-3","volume":"52","author":"C Wang","year":"2021","unstructured":"Wang, C., Gu, X.: Image captioning with adaptive incremental global context attention. Appl. Intell. 52, 6575\u20136597 (2021)","journal-title":"Appl. Intell."},{"key":"3905_CR79","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2023.109420","volume":"138","author":"Y Ma","year":"2023","unstructured":"Ma, Y., Ji, J., Sun, X., Zhou, Y., Ji, R.: Towards local visual modeling for image captioning. Pattern Recognit. 138, 109420 (2023)","journal-title":"Pattern Recognit."},{"key":"3905_CR80","doi-asserted-by":"publisher","first-page":"92","DOI":"10.1109\/TMM.2020.2976552","volume":"23","author":"J Zhang","year":"2021","unstructured":"Zhang, J., Mei, K., Zheng, Y., Fan, J.: Integrating part of speech guidance for image captioning. IEEE Trans. Multimed. 23, 92\u2013104 (2021)","journal-title":"IEEE Trans. Multimed."}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-03905-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-025-03905-w\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-03905-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,6]],"date-time":"2025-09-06T13:41:25Z","timestamp":1757166085000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-025-03905-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,8]]},"references-count":80,"journal-issue":{"issue":"11","published-print":{"date-parts":[[2025,9]]}},"alternative-id":["3905"],"URL":"https:\/\/doi.org\/10.1007\/s00371-025-03905-w","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"value":"0178-2789","type":"print"},{"value":"1432-2315","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,5,8]]},"assertion":[{"value":"24 March 2025","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 May 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}