{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,23]],"date-time":"2026-06-23T07:51:48Z","timestamp":1782201108291,"version":"3.54.5"},"reference-count":92,"publisher":"Springer Science and Business Media LLC","issue":"8","license":[{"start":{"date-parts":[[2026,5,20]],"date-time":"2026-05-20T00:00:00Z","timestamp":1779235200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,5,20]],"date-time":"2026-05-20T00:00:00Z","timestamp":1779235200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Shanghai Municipal Science and Technology Major Project","award":["2021SHZDZX0100"],"award-info":[{"award-number":["2021SHZDZX0100"]}]},{"DOI":"10.13039\/501100001809","name":"National Nature Science Foundation of China","doi-asserted-by":"crossref","award":["62273256"],"award-info":[{"award-number":["62273256"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"crossref","id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1007\/s00371-026-04509-8","type":"journal-article","created":{"date-parts":[[2026,5,20]],"date-time":"2026-05-20T08:20:34Z","timestamp":1779265234000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["RGFRCap: enhancing image captioning with retrieval-guided semantic feature refinement"],"prefix":"10.1007","volume":"42","author":[{"given":"Jiaqi","family":"Fan","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hongqing","family":"Chu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hao","family":"Fang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jia","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Quanbo","family":"Ge","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bingzhao","family":"Gao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,5,20]]},"reference":[{"key":"4509_CR1","doi-asserted-by":"publisher","DOI":"10.21203\/rs.3.rs-5169088\/v1","author":"Y Zhang","year":"2025","unstructured":"Zhang, Y., Tong, J., Liu, H.: SCAP: enhancing image captioning through lightweight feature sifting and hierarchical decoding. Visual Comput (2025). https:\/\/doi.org\/10.21203\/rs.3.rs-5169088\/v1","journal-title":"Visual Comput"},{"key":"4509_CR2","doi-asserted-by":"publisher","first-page":"8895","DOI":"10.1007\/s00371-025-03905-w","volume":"41","author":"J Yang","year":"2025","unstructured":"Yang, J., Zhang, H., Wang, R., Xue, L., Zhang, J.: Semantically enhanced dual visual fusion transformer for accurate image captioning. Visual Comput 41, 8895\u20138910 (2025)","journal-title":"Visual Comput"},{"key":"4509_CR3","doi-asserted-by":"publisher","first-page":"10841","DOI":"10.1007\/s00371-025-04072-8","volume":"41","author":"BT Hung","year":"2025","unstructured":"Hung, B.T., Huy, V.Q.: MTFIC: enhanced fashion image captioning via multi-transformer architecture with contrastive and bidirectional encodings. Visual Comput 41, 10841\u201310855 (2025)","journal-title":"Visual Comput"},{"key":"4509_CR4","doi-asserted-by":"crossref","unstructured":"Wen, Y., Luo, B., Shi, W., Ji, J., Cao, W., Yang, X., Sheng, B.: SAT-Net: structure-aware transformer-based attention fusion network for low-quality retinal fundus images enhancement. IEEE Trans. Multimedia (2025)","DOI":"10.1109\/TMM.2025.3565935"},{"key":"4509_CR5","doi-asserted-by":"publisher","first-page":"2226","DOI":"10.1109\/TMM.2022.3144890","volume":"25","author":"N Jiang","year":"2022","unstructured":"Jiang, N., Sheng, B., Li, P., Lee, T.-Y.: Photohelper: portrait photographing guidance via deep feature retrieval and fusion. IEEE Trans. Multimedia 25, 2226\u20132238 (2022)","journal-title":"IEEE Trans. Multimedia"},{"issue":"1","key":"4509_CR6","doi-asserted-by":"publisher","first-page":"9","DOI":"10.1007\/s00371-021-02309-w","volume":"39","author":"B Sun","year":"2023","unstructured":"Sun, B., Wu, Y., Zhao, Y., Hao, Z., Yu, L., He, J.: Cross-language multimodal scene semantic guidance and leap sampling for video captioning. Vis. Comput. 39(1), 9\u201325 (2023)","journal-title":"Vis. Comput."},{"key":"4509_CR7","doi-asserted-by":"publisher","first-page":"5405","DOI":"10.1007\/s00371-024-03729-0","volume":"41","author":"F Liang","year":"2024","unstructured":"Liang, F., Huang, Z., Wang, W., He, Z., En, Q.: Dynamic text prompt joint multimodal features for accurate plant disease image captioning. Visual Comput 41, 5405\u20135419 (2024)","journal-title":"Visual Comput"},{"issue":"3","key":"4509_CR8","doi-asserted-by":"publisher","first-page":"2267","DOI":"10.1002\/cav.2267","volume":"35","author":"P Zhou","year":"2024","unstructured":"Zhou, P., Li, C., Zhang, J., Wang, C., Qin, H., Liu, L.: A novel transformer-based graph generation model for vectorized road design. Comput Anim Virtual Worlds 35(3), 2267 (2024)","journal-title":"Comput Anim Virtual Worlds"},{"key":"4509_CR9","doi-asserted-by":"publisher","first-page":"3615","DOI":"10.1109\/TITS.2023.3323085","volume":"25","author":"C Liu","year":"2023","unstructured":"Liu, C., Zhang, X., Chang, F., Li, S., Hao, P., Lu, Y., Wang, Y.: Traffic scenario understanding and video captioning via guidance attention captioning network. IEEE Trans. Intell. Transp. Syst. 25, 3615\u20133627 (2023)","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"4509_CR10","doi-asserted-by":"crossref","unstructured":"Vinyals, O., Toshev, A., Bengio, S., Erhan, D.: Show and tell: a neural image caption generator. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3156\u20133164 (2015)","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"4509_CR11","doi-asserted-by":"crossref","unstructured":"Wu, Q., Shen, C., Liu, L., Dick, A., Van Den\u00a0Hengel, A.: What value do explicit high level concepts have in vision to language problems? In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 203\u2013212 (2016)","DOI":"10.1109\/CVPR.2016.29"},{"key":"4509_CR12","doi-asserted-by":"crossref","unstructured":"Anderson, P., He, X., Buehler, C., Teney, D., Johnson, M., Gould, S., Zhang, L.: Bottom-up and top-down attention for image captioning and visual question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 6077\u20136086 (2018)","DOI":"10.1109\/CVPR.2018.00636"},{"key":"4509_CR13","doi-asserted-by":"crossref","unstructured":"Cornia, M., Stefanini, M., Baraldi, L., Cucchiara, R.: Meshed-memory transformer for image captioning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 10578\u201310587 (2020)","DOI":"10.1109\/CVPR42600.2020.01059"},{"key":"4509_CR14","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2024.110555","volume":"153","author":"S Cao","year":"2024","unstructured":"Cao, S., An, G., Cen, Y., Yang, Z., Lin, W.: Cast: cross-modal retrieval and visual conditioning for image captioning. Pattern Recogn. 153, 110555 (2024)","journal-title":"Pattern Recogn."},{"issue":"5","key":"4509_CR15","doi-asserted-by":"publisher","first-page":"2293","DOI":"10.1002\/cav.2293","volume":"35","author":"Z Xiao","year":"2024","unstructured":"Xiao, Z., Chen, Y., Zhou, X., He, M., Liu, L., Yu, F., Jiang, M.: Human action recognition in immersive virtual reality based on multi-scale spatio-temporal attention network. Comput Anim Virtual Worlds 35(5), 2293 (2024)","journal-title":"Comput Anim Virtual Worlds"},{"key":"4509_CR16","unstructured":"Li, J., Li, D., Savarese, S., Hoi, S.: Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. In: Proceedings of the 36th International Conference on Machine Learning, (ICML), pp. 19730\u201319742 (2023)"},{"key":"4509_CR17","unstructured":"Chen, X., Wang, X., Changpinyo, S., Piergiovanni, A., Padlewski, P., Salz, D., Goodman, S., Grycner, A., Mustafa, B., Beyer, L., et al.: PaLI: a jointly-scaled multilingual language-image model. arXiv preprint arXiv:2209.06794 (2022)"},{"key":"4509_CR18","doi-asserted-by":"crossref","unstructured":"Ramos, R., Martins, B., Elliott, D., Kementchedjhieva, Y.: Smallcap: lightweight image captioning prompted with retrieval augmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 2840\u20132849 (2023)","DOI":"10.1109\/CVPR52729.2023.00278"},{"key":"4509_CR19","unstructured":"Luo, Z., Xi, Y., Zhang, R., Ma, J.: I-tuning: tuning language models with image for caption generation. arXiv preprint arXiv:2202.06574 (2022)"},{"key":"4509_CR20","doi-asserted-by":"crossref","unstructured":"Yang, Z., Ping, W., Liu, Z., Korthikanti, V., Nie, W., Huang, D.-A., Fan, L., Yu, Z., Lan, S., Li, B., et al.: Re-ViLM: retrieval-augmented visual language model for zero and few-shot image captioning. arXiv preprint arXiv:2302.04858 (2023)","DOI":"10.18653\/v1\/2023.findings-emnlp.793"},{"key":"4509_CR21","doi-asserted-by":"crossref","unstructured":"Sarto, S., Cornia, M., Baraldi, L., Nicolosi, A., Cucchiara, R.: Towards retrieval-augmented architectures for image captioning. In: ACM Transactions on Multimedia Computing, Communications and Applications (2024)","DOI":"10.1145\/3663667"},{"key":"4509_CR22","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2024.125847","volume":"264","author":"N Gao","year":"2025","unstructured":"Gao, N., Yao, R., Chen, P., Liang, R., Sun, G., Tang, J.: Multi-granularity semantic relational mapping for image caption. Expert Syst. Appl. 264, 125847 (2025)","journal-title":"Expert Syst. Appl."},{"key":"4509_CR23","doi-asserted-by":"crossref","unstructured":"Sarto, S., Cornia, M., Baraldi, L., Cucchiara, R.: Retrieval-augmented transformer for image captioning. In: Proceedings of the 19th International Conference on Content-Based Multimedia Indexing, pp. 1\u20137 (2022)","DOI":"10.1145\/3549555.3549585"},{"key":"4509_CR24","doi-asserted-by":"crossref","unstructured":"Ramos, R., Elliott, D., Martins, B.: Retrieval-augmented image captioning. arXiv preprint arXiv:2302.08268 (2023)","DOI":"10.18653\/v1\/2023.eacl-main.266"},{"issue":"8","key":"4509_CR25","doi-asserted-by":"publisher","first-page":"9731","DOI":"10.1007\/s10489-022-04010-4","volume":"53","author":"S Zhao","year":"2023","unstructured":"Zhao, S., Li, L., Peng, H.: Incorporating retrieval-based method for feature enhanced image captioning. Appl. Intell. 53(8), 9731\u20139743 (2023)","journal-title":"Appl. Intell."},{"key":"4509_CR26","doi-asserted-by":"crossref","unstructured":"Kim, T., Lee, S., Kim, S.-W., Kim, D.-J.: ViPCap: retrieval text-based visual prompts for lightweight image captioning. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 39, pp. 4320\u20134328 (2025)","DOI":"10.1609\/aaai.v39i4.32454"},{"key":"4509_CR27","doi-asserted-by":"crossref","unstructured":"Xiao, S., Li, B., Siyu, J., Gengqi, Y., Zisen, Q., Dayan, W., Haiping, W., Yu, D.: HRACap: lightweight image captioning via hierarchical retrieval-augmented prompt. In: International Conference on Intelligent Computing, pp. 3\u201315 (2025)","DOI":"10.1007\/978-981-96-9961-2_1"},{"key":"4509_CR28","doi-asserted-by":"crossref","unstructured":"Li, Y., Pan, Y., Yao, T., Mei, T.: Comprehending and ordering semantics for image captioning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 17990\u201317999 (2022)","DOI":"10.1109\/CVPR52688.2022.01746"},{"key":"4509_CR29","doi-asserted-by":"crossref","unstructured":"Huang, L., Wang, W., Chen, J., Wei, X.-Y.: Attention on attention for image captioning. In: Proceedings of the IEEE International Conference on Computer Vision (ICCV), pp. 4634\u20134643 (2019)","DOI":"10.1109\/ICCV.2019.00473"},{"key":"4509_CR30","doi-asserted-by":"crossref","unstructured":"Rennie, S.J., Marcheret, E., Mroueh, Y., Ross, J., Goel, V.: Self-critical sequence training for image captioning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 7008\u20137024 (2017)","DOI":"10.1109\/CVPR.2017.131"},{"key":"4509_CR31","doi-asserted-by":"crossref","unstructured":"Chen, L., Zhang, H., Xiao, J., Nie, L., Shao, J., Liu, W., Chua, T.-S.: SCA-CNN: spatial and channel-wise attention in convolutional networks for image captioning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 5659\u20135667 (2017)","DOI":"10.1109\/CVPR.2017.667"},{"key":"4509_CR32","unstructured":"Xu, K., Ba, J., Kiros, R., Cho, K., Courville, A., Salakhudinov, R., Zemel, R., Bengio, Y.: Show, attend and tell: neural image caption generation with visual attention. In: Proceedings of the International Conference on Machine Learning (ICML), pp. 2048\u20132057 (2015)"},{"key":"4509_CR33","doi-asserted-by":"crossref","unstructured":"Yao, T., Pan, Y., Li, Y., Mei, T.: Exploring visual relationship for image captioning. In: Proc. Eur. Conf. Comput. Vis. (ECCV), pp. 684\u2013699 (2018)","DOI":"10.1007\/978-3-030-01264-9_42"},{"key":"4509_CR34","doi-asserted-by":"publisher","DOI":"10.1016\/j.compeleceng.2024.109626","volume":"119","author":"F Xiao","year":"2024","unstructured":"Xiao, F., Zhang, N., Xue, W., Gao, X.: Sentinel mechanism for visual semantic graph-based image captioning. Comput. Electr. Eng. 119, 109626 (2024)","journal-title":"Comput. Electr. Eng."},{"key":"4509_CR35","unstructured":"Herdade, S., Kappeler, A., Boakye, K., Soares, J.: Image captioning: transforming objects into words. In: Proceeding of the Advances in Neural Information Processing Systems (NeurIPS), pp. 11135\u201311145 (2019)"},{"issue":"1","key":"4509_CR36","doi-asserted-by":"publisher","first-page":"48","DOI":"10.1007\/s00530-023-01230-7","volume":"30","author":"P Yan","year":"2024","unstructured":"Yan, P., Li, Z., Hu, R., Cao, X.: BENet: bi-directional enhanced network for image captioning. Multimedia Syst. 30(1), 48 (2024)","journal-title":"Multimedia Syst."},{"issue":"6","key":"4509_CR37","doi-asserted-by":"publisher","first-page":"4794","DOI":"10.1007\/s10489-024-05416-y","volume":"54","author":"Q Su","year":"2024","unstructured":"Su, Q., Hu, J., Li, Z.: Visual contextual relationship augmented transformer for image captioning. Appl. Intell. 54(6), 4794\u20134813 (2024)","journal-title":"Appl. Intell."},{"key":"4509_CR38","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2024.123847","volume":"250","author":"X Yang","year":"2024","unstructured":"Yang, X., Yang, Y., Wu, J., Sun, W., Ma, S., Hou, Z.: CA-Captioner: a novel concentrated attention for image captioning. Expert Syst. Appl. 250, 123847 (2024)","journal-title":"Expert Syst. Appl."},{"key":"4509_CR39","doi-asserted-by":"crossref","unstructured":"Luo, J., Li, Y., Pan, Y., Yao, T., Feng, J., Chao, H., Mei, T.: Semantic-conditional diffusion networks for image captioning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition(CVPR), pp. 23359\u201323368 (2023)","DOI":"10.1109\/CVPR52729.2023.02237"},{"key":"4509_CR40","doi-asserted-by":"crossref","unstructured":"Guo, L., Liu, J., Zhu, X., Yao, P., Lu, S., Lu, H.: Normalized and geometry-aware self-attention network for image captioning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 10327\u201310336 (2020)","DOI":"10.1109\/CVPR42600.2020.01034"},{"key":"4509_CR41","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2022.117174","volume":"201","author":"C Wang","year":"2022","unstructured":"Wang, C., Shen, Y., Ji, L.: Geometry attention transformer with position-aware LSTMS for image captioning. Expert Syst. Appl. 201, 117174 (2022)","journal-title":"Expert Syst. Appl."},{"key":"4509_CR42","doi-asserted-by":"crossref","unstructured":"Wu, M., Zhang, X., Sun, X., Zhou, Y., Chen, C., Gu, J., Sun, X., Ji, R.: DIFNet: boosting visual information flow for image captioning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 18020\u201318029 (2022)","DOI":"10.1109\/CVPR52688.2022.01749"},{"issue":"5","key":"4509_CR43","doi-asserted-by":"publisher","first-page":"118","DOI":"10.1007\/s00138-024-01599-z","volume":"35","author":"DC Bui","year":"2024","unstructured":"Bui, D.C., Nguyen, T.V., Nguyen, K.: Transformer with multi-level grid features and depth pooling for image captioning. Mach. Vis. Appl. 35(5), 118 (2024)","journal-title":"Mach. Vis. Appl."},{"key":"4509_CR44","doi-asserted-by":"publisher","DOI":"10.1016\/j.engappai.2023.106384","volume":"123","author":"J Hu","year":"2023","unstructured":"Hu, J., Yang, Y., An, Y., Yao, L.: Dual-spatial normalized transformer for image captioning. Eng. Appl. Artif. Intell. 123, 106384 (2023)","journal-title":"Eng. Appl. Artif. Intell."},{"key":"4509_CR45","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2024.110941","volume":"158","author":"Y Li","year":"2025","unstructured":"Li, Y., Ji, J., Sun, X., Zhou, Y., Luo, Y., Ji, R.: M3ixup: a multi-modal data augmentation approach for image captioning. Pattern Recogn. 158, 110941 (2025)","journal-title":"Pattern Recogn."},{"key":"4509_CR46","doi-asserted-by":"crossref","unstructured":"Xu, H., Yan, M., Li, C., Bi, B., Huang, S., Xiao, W., Huang, F.: E2E-VLP: end-to-end vision-language pre-training enhanced by visual learning. arXiv preprint arXiv:2106.01804 (2021)","DOI":"10.18653\/v1\/2021.acl-long.42"},{"key":"4509_CR47","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2025.113127","volume":"311","author":"W Zhou","year":"2025","unstructured":"Zhou, W., Song, C., Chen, D., Su, T., Hu, H., Shan, C.: From multi-scale grids to dynamic regions: dual-relation enhanced transformer for image captioning. Knowl. Based Syst. 311, 113127 (2025)","journal-title":"Knowl. Based Syst."},{"key":"4509_CR48","unstructured":"Mokady, R., Hertz, A., Bermano, A.H.: ClipCap: clip prefix for image captioning. arXiv preprint arXiv:2111.09734 (2021)"},{"key":"4509_CR49","unstructured":"Wang, J., Hu, X., Zhang, P., Li, X., Wang, L., Zhang, L., Gao, J., Liu, Z.: MiniVLM: a smaller and faster vision-language model. arXiv preprint arXiv:2012.06946 (2020)"},{"key":"4509_CR50","doi-asserted-by":"crossref","unstructured":"Fang, Z., Wang, J., Hu, X., Wang, L., Yang, Y., Liu, Z.: Compressing visual-linguistic model via knowledge distillation. In: Proceedings of the IEEE International Conference on Computer Vision (ICCV), pp. 1428\u20131438 (2021)","DOI":"10.1109\/ICCV48922.2021.00146"},{"issue":"6","key":"4509_CR51","doi-asserted-by":"publisher","first-page":"509","DOI":"10.1016\/j.vrih.2023.06.003","volume":"5","author":"M Wang","year":"2023","unstructured":"Wang, M., Meng, M., Liu, J., Wu, J.: Adequate alignment and interaction for cross-modal retrieval. Virtual Real. Intell. Hardware 5(6), 509\u2013522 (2023)","journal-title":"Virtual Real. Intell. Hardware"},{"key":"4509_CR52","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., et al.: Learning transferable visual models from natural language supervision. In: Proceedings of the International Conference on Machine Learning (ICML), pp. 8748\u20138763 (2021)"},{"key":"4509_CR53","doi-asserted-by":"crossref","unstructured":"Hu, Z., Iscen, A., Sun, C., Wang, Z., Chang, K.-W., Sun, Y., Schmid, C., Ross, D.A., Fathi, A.: Reveal: retrieval-augmented visual-language pre-training with multi-source multimodal knowledge memory. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 23369\u201323379 (2023)","DOI":"10.1109\/CVPR52729.2023.02238"},{"key":"4509_CR54","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster R-CNN: towards real-time object detection with region proposal networks. In: Proceedings of the Advances in Neural Information Processing Systems (NeurIPS), pp. 91\u201399 (2015)"},{"key":"4509_CR55","doi-asserted-by":"crossref","unstructured":"Chen, L.-C., Zhu, Y., Papandreou, G., Schroff, F., Adam, H.: Encoder-decoder with atrous separable convolution for semantic image segmentation. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 801\u2013818 (2018)","DOI":"10.1007\/978-3-030-01234-2_49"},{"key":"4509_CR56","unstructured":"Mikolov, T., Chen, K., Corrado, G., Dean, J.: Efficient estimation of word representations in vector space. arXiv preprint arXiv:1301.3781 (2013)"},{"key":"4509_CR57","doi-asserted-by":"crossref","unstructured":"Hu, J., Shen, L., Sun, G.: Squeeze-and-excitation networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 7132\u20137141 (2018)","DOI":"10.1109\/CVPR.2018.00745"},{"key":"4509_CR58","doi-asserted-by":"crossref","unstructured":"Woo, S., Park, J., Lee, J.-Y., Kweon, I.S.: CBAM: convolutional block attention module. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 3\u201319 (2018)","DOI":"10.1007\/978-3-030-01234-2_1"},{"key":"4509_CR59","doi-asserted-by":"crossref","unstructured":"Fu, J., Liu, J., Tian, H., Li, Y., Bao, Y., Fang, Z., Lu, H.: Dual attention network for scene segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3146\u20133154 (2019)","DOI":"10.1109\/CVPR.2019.00326"},{"key":"4509_CR60","doi-asserted-by":"crossref","unstructured":"Hou, Q., Zhou, D., Feng, J.: Coordinate attention for efficient mobile network design. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 13713\u201313722 (2021)","DOI":"10.1109\/CVPR46437.2021.01350"},{"key":"4509_CR61","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r, P., Zitnick, C.L.: Microsoft coco: common objects in context. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 740\u2013755 (2014)","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"4509_CR62","doi-asserted-by":"crossref","unstructured":"Plummer, B.A., Wang, L., Cervantes, C.M., Caicedo, J.C., Hockenmaier, J., Lazebnik, S.: Flickr30k entities: collecting region-to-phrase correspondences for richer image-to-sentence models. In: Proceedings of the IEEE International Conference on Computer Vision (ICCV), pp. 2641\u20132649 (2015)","DOI":"10.1109\/ICCV.2015.303"},{"key":"4509_CR63","doi-asserted-by":"crossref","unstructured":"Cordts, M., Omran, M., Ramos, S., Rehfeld, T., Enzweiler, M., Benenson, R., Franke, U., Roth, S., Schiele, B.: The cityscapes dataset for semantic urban scene understanding. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3213\u20133223 (2016)","DOI":"10.1109\/CVPR.2016.350"},{"key":"4509_CR64","doi-asserted-by":"publisher","first-page":"973","DOI":"10.1007\/s11263-018-1072-8","volume":"126","author":"C Sakaridis","year":"2018","unstructured":"Sakaridis, C., Dai, D., Van Gool, L.: Semantic foggy scene understanding with synthetic data. Int. J. Comput. Vis. 126, 973\u2013992 (2018)","journal-title":"Int. J. Comput. Vis."},{"key":"4509_CR65","doi-asserted-by":"crossref","unstructured":"Papineni, K., Roukos, S., Ward, T., Zhu, W.-J.: Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the Annual Meeting of the Association for Computational Linguistics, pp. 311\u2013318 (2002)","DOI":"10.3115\/1073083.1073135"},{"key":"4509_CR66","unstructured":"Banerjee, S., Lavie, A.: Meteor: an automatic metric for MT evaluation with improved correlation with human judgments. In: Proceedings of the AC Workshop on Intrinsic and Extrinsic Evaluation Measures for Machine Translation And\/or Summarization, pp. 65\u201372 (2005)"},{"key":"4509_CR67","unstructured":"Lin, C.-Y.: Rouge: a package for automatic evaluation of summaries. In: Proceedings of Workshop on Text Summarization Branches Out, pp. 74\u201381 (2004)"},{"key":"4509_CR68","doi-asserted-by":"crossref","unstructured":"Vedantam, R., Lawrence\u00a0Zitnick, C., Parikh, D.: Cider: consensus-based image description evaluation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 4566\u20134575 (2015)","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"4509_CR69","doi-asserted-by":"crossref","unstructured":"Anderson, P., Fernando, B., Johnson, M., Gould, S.: Spice: semantic propositional image caption evaluation. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 382\u2013398 (2016)","DOI":"10.1007\/978-3-319-46454-1_24"},{"issue":"21","key":"4509_CR70","doi-asserted-by":"publisher","first-page":"10748","DOI":"10.1007\/s10489-024-05739-w","volume":"54","author":"A Mundu","year":"2024","unstructured":"Mundu, A., Singh, S.K., Dubey, S.R.: ETransCap: efficient transformer for image captioning. Appl. Intell. 54(21), 10748\u201310762 (2024)","journal-title":"Appl. Intell."},{"key":"4509_CR71","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2025.128597","volume":"292","author":"C Shan","year":"2025","unstructured":"Shan, C., Song, C., Zou, T., Li, J., Liu, S.: Dual dynamic transformer for image captioning. Expert Syst. Appl. 292, 128597 (2025)","journal-title":"Expert Syst. Appl."},{"key":"4509_CR72","first-page":"9809","volume":"35","author":"J Li","year":"2023","unstructured":"Li, J., Zhang, L., Zhang, K., Hu, B., Xie, H., Mao, Z.: Cascade semantic prompt alignment network for image captioning. IEEE Trans. Circuits Syst. Video Technol. 35, 9809\u20139822 (2023)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"4509_CR73","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2025.126850","volume":"273","author":"W Zhou","year":"2025","unstructured":"Zhou, W., Jiang, W., Zheng, Z., Li, J., Su, T., Hu, H.: From grids to pseudo-regions: dynamic memory augmented image captioning with dual relation transformer. Expert Syst. Appl. 273, 126850 (2025)","journal-title":"Expert Syst. Appl."},{"key":"4509_CR74","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2024.127823","volume":"593","author":"X Yang","year":"2024","unstructured":"Yang, X., Yang, Y., Ma, S., Li, Z., Dong, W., Wo\u017aniak, M.: SAMT-generator: a second-attention for image captioning based on multi-stage transformer network. Neurocomputing 593, 127823 (2024)","journal-title":"Neurocomputing"},{"key":"4509_CR75","doi-asserted-by":"publisher","first-page":"11900","DOI":"10.1109\/TCSVT.2024.3425513","volume":"34","author":"L Wang","year":"2024","unstructured":"Wang, L., Chen, H., Liu, Y., Lyu, Y.: Regular constrained multimodal fusion for image captioning. IEEE Trans. Circuits Syst. Video Technol. 34, 11900\u201311913 (2024)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"4509_CR76","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2025.126692","volume":"272","author":"MS Hossain","year":"2025","unstructured":"Hossain, M.S., Aktar, S., Gu, N., Liu, W., Huang, Z.: GEOSCN: a novel multimodal self-attention to integrate geometric information on spatial-channel network for fine-grained image captioning. Expert Syst. Appl. 272, 126692 (2025)","journal-title":"Expert Syst. Appl."},{"key":"4509_CR77","doi-asserted-by":"crossref","unstructured":"Jaiswal, T., Pandey, M., Tripathi, P.: Advancing image captioning with V16HP1365 encoder and dual self-attention network. In: Multimedia Tools and Applications, pp. 1\u201325 (2024)","DOI":"10.1007\/s11042-024-18467-7"},{"issue":"2","key":"4509_CR78","doi-asserted-by":"publisher","first-page":"4219","DOI":"10.1007\/s11042-023-15291-3","volume":"83","author":"D Sharma","year":"2024","unstructured":"Sharma, D., Dhiman, C., Kumar, D.: XGL-T transformer model for intelligent image captioning. Multimed. Tools Appl. 83(2), 4219\u20134240 (2024)","journal-title":"Multimed. Tools Appl."},{"key":"4509_CR79","doi-asserted-by":"crossref","unstructured":"Zhang, P., Li, X., Hu, X., Yang, J., Zhang, L., Wang, L., Choi, Y., Gao, J.: VinVL: revisiting visual representations in vision-language models. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 5579\u20135588 (2021)","DOI":"10.1109\/CVPR46437.2021.00553"},{"key":"4509_CR80","doi-asserted-by":"crossref","unstructured":"Wang, W., Lv, Q., Yu, W., Hong, W., Qi, J., Wang, Y., Ji, J., Yang, Z., Zhao, L., Song, X., et al.: CogVLM: visual expert for pretrained language models. arXiv preprint arXiv:2311.03079 (2023)","DOI":"10.52202\/079017-3860"},{"key":"4509_CR81","doi-asserted-by":"crossref","unstructured":"Chen, X., Djolonga, J., Padlewski, P., Mustafa, B., Changpinyo, S., Wu, J., Ruiz, C.R., Goodman, S., Wang, X., Tay, Y., et al.: PaLI-X: on scaling up a multilingual vision and language model. arXiv preprint arXiv:2305.18565 (2023)","DOI":"10.1109\/CVPR52733.2024.01368"},{"key":"4509_CR82","doi-asserted-by":"crossref","unstructured":"Barraco, M., Stefanini, M., Cornia, M., Cascianelli, S., Baraldi, L., Cucchiara, R.: CaMEL: mean teacher learning for image captioning. In: 2022 26th International Conference on Pattern Recognition (ICPR), pp. 4087\u20134094 (2022)","DOI":"10.1109\/ICPR56361.2022.9955644"},{"key":"4509_CR83","doi-asserted-by":"crossref","unstructured":"Lu, J., Xiong, C., Parikh, D., Socher, R.: Knowing when to look: adaptive attention via a visual sentinel for image captioning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 375\u2013383 (2017)","DOI":"10.1109\/CVPR.2017.345"},{"key":"4509_CR84","doi-asserted-by":"publisher","first-page":"43","DOI":"10.1016\/j.patrec.2020.12.020","volume":"143","author":"Y Zhang","year":"2021","unstructured":"Zhang, Y., Shi, X., Mi, S., Yang, X.: Image captioning with transformer and knowledge graph. Pattern Recogn. Lett. 143, 43\u201349 (2021)","journal-title":"Pattern Recogn. Lett."},{"key":"4509_CR85","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2023.109420","volume":"138","author":"Y Ma","year":"2023","unstructured":"Ma, Y., Ji, J., Sun, X., Zhou, Y., Ji, R.: Towards local visual modeling for image captioning. Pattern Recogn. 138, 109420 (2023)","journal-title":"Pattern Recogn."},{"issue":"9","key":"4509_CR86","doi-asserted-by":"publisher","first-page":"6561","DOI":"10.1007\/s13042-025-02634-9","volume":"16","author":"R Wang","year":"2025","unstructured":"Wang, R., Li, S., Xue, L., Yang, J.: Enhancing visual contextual semantic information for image captioning. Int. J. Mach. Learn. Cybern. 16(9), 6561\u20136576 (2025)","journal-title":"Int. J. Mach. Learn. Cybern."},{"key":"4509_CR87","doi-asserted-by":"publisher","DOI":"10.1016\/j.engappai.2025.110330","volume":"147","author":"W Zhu","year":"2025","unstructured":"Zhu, W., Jiang, Z., He, Y.: Geometry-sensitive semantic modeling in visual and visual-language domains for image captioning. Eng. Appl. Artif. Intell. 147, 110330 (2025)","journal-title":"Eng. Appl. Artif. Intell."},{"key":"4509_CR88","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2025.130622","volume":"647","author":"W Li","year":"2025","unstructured":"Li, W., Li, X., Li, Z., Yu, J., Chen, P.: Hi-captioner: end-to-end image captioning based on hierarchical multi-scale encoding and cross-modal interactive decoding. Neurocomputing 647, 130622 (2025)","journal-title":"Neurocomputing"},{"key":"4509_CR89","doi-asserted-by":"crossref","unstructured":"He, S., Liao, W., Tavakoli, H.R., Yang, M., Rosenhahn, B., Pugeault, N.: Image captioning through image transformer. In: Proceedings of the Asian Conference on Computer Vision (ACCV), pp. 153\u2013169 (2020)","DOI":"10.1007\/978-3-030-69538-5_10"},{"key":"4509_CR90","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, \u0141., Polosukhin, I.: Attention is all you need. In: Proceedings of the Advances in Neural Information Processing Systems (NeurIPS), pp. 5998\u20136008 (2017)"},{"issue":"8","key":"4509_CR91","doi-asserted-by":"publisher","first-page":"4257","DOI":"10.1109\/TCSVT.2023.3243725","volume":"33","author":"J Zhang","year":"2024","unstructured":"Zhang, J., Xie, Y., Ding, W., Wang, Z.: Cross on cross attention: deep fusion transformer for image captioning. IEEE Trans. Circuits Syst. Video Technol. 33(8), 4257\u20134268 (2024)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"4509_CR92","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2025.107390","volume":"187","author":"Q Xu","year":"2025","unstructured":"Xu, Q., Song, S., Wu, Q., Jiang, B., Luo, B., Tang, J.: Multi-level semantic-aware transformer for image captioning. Neural Netw. 187, 107390 (2025)","journal-title":"Neural Netw."}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-026-04509-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-026-04509-8","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-026-04509-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,23]],"date-time":"2026-06-23T07:42:24Z","timestamp":1782200544000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-026-04509-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,20]]},"references-count":92,"journal-issue":{"issue":"8","published-print":{"date-parts":[[2026,6]]}},"alternative-id":["4509"],"URL":"https:\/\/doi.org\/10.1007\/s00371-026-04509-8","relation":{"has-preprint":[{"id-type":"doi","id":"10.21203\/rs.3.rs-7721105\/v1","asserted-by":"object"}]},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"value":"0178-2789","type":"print"},{"value":"1432-2315","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,5,20]]},"assertion":[{"value":"26 September 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 April 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 May 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"All authors agreed with the content and gave explicit consent to submit.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}}],"article-number":"305"}}