{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,7]],"date-time":"2026-06-07T04:20:23Z","timestamp":1780806023624,"version":"3.54.1"},"reference-count":45,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2022,12,19]],"date-time":"2022-12-19T00:00:00Z","timestamp":1671408000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,12,19]],"date-time":"2022-12-19T00:00:00Z","timestamp":1671408000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2023,6]]},"DOI":"10.1007\/s00530-022-01036-z","type":"journal-article","created":{"date-parts":[[2022,12,19]],"date-time":"2022-12-19T21:33:10Z","timestamp":1671485590000},"page":"1043-1056","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":11,"title":["Layer-wise enhanced transformer with multi-modal fusion for image caption"],"prefix":"10.1007","volume":"29","author":[{"given":"Jingdan","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yi","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dexin","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2022,12,19]]},"reference":[{"key":"1036_CR1","doi-asserted-by":"crossref","unstructured":"Anderson, P., Fernando, B., Johnson, M., Gould, S.: (2016) SPICE: semantic propositional image caption evaluation. In: Computer Vision\u2014ECCV 2016\u201414th European Conference, Amsterdam, The Netherlands, October 11\u201314, 2016, Proceedings, Part V","DOI":"10.1007\/978-3-319-46454-1_24"},{"key":"1036_CR2","doi-asserted-by":"crossref","unstructured":"Anderson, P., He, X., Buehler, C., Teney, D., Johnson, M., Gould, S., Zhang, L.: Bottom-up and top-down attention for image captioning and VQA. CoRR:abs\/1707.07998 (2017)","DOI":"10.1109\/CVPR.2018.00636"},{"key":"1036_CR3","unstructured":"Ba, L.J., Kiros, J.R., Hinton, G.E.: Layer normalization. CoRR:abs\/1607.06450 (2016)"},{"key":"1036_CR4","doi-asserted-by":"crossref","unstructured":"Biten, AF., Litman, R., Xie, Y., Appalaraju, S., Manmatha, R.: Latr: layout-aware transformer for scene-text VQA. CoRR abs\/2112.12494, 2112.12494 (2021)","DOI":"10.1109\/CVPR52688.2022.01605"},{"key":"1036_CR5","unstructured":"Chung, J., G\u00fcl\u00e7ehre, \u00c7., Cho, K., Bengio, Y.: Empirical evaluation of gated recurrent neural networks on sequence modeling. CoRR abs\/1412.3555 (2014)"},{"key":"1036_CR6","doi-asserted-by":"crossref","unstructured":"Cornia, M., Stefanini, M., Baraldi, L., Cucchiara, R.: Meshed-memory transformer for image captioning. In: 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2020, Seattle, WA, USA, June 13\u201319, 2020, Computer Vision Foundation\/IEEE, pp. 10575\u201310584 (2020)","DOI":"10.1109\/CVPR42600.2020.01059"},{"key":"1036_CR7","unstructured":"Devlin, J., Chang, M., Lee, K., Toutanova, K.: BERT: pre-training of deep bidirectional transformers for language understanding. Association for Computational Linguistics, pp. 4171\u20134186 (2019)"},{"key":"1036_CR8","doi-asserted-by":"crossref","unstructured":"Fang, H., Gupta, S., Iandola, F.N., Srivastava, R.K., Deng, L., Doll\u00e1r, P., Gao, J., He, X., Mitchell, M., Platt, J.C., Zitnick, C.L., Zweig, G.: From captions to visual concepts and back. IEEE Computer Society, pp. 1473\u20131482 (2015)","DOI":"10.1109\/CVPR.2015.7298754"},{"key":"1036_CR9","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. CoRR abs\/1512.03385 (2015)","DOI":"10.1109\/CVPR.2016.90"},{"key":"1036_CR10","unstructured":"Herdade, S., Kappeler, A., Boakye, K., Soares, J.: (2019) Image captioning: transforming objects into words. pp. 11135\u201311145"},{"key":"1036_CR11","first-page":"1693","volume":"28","author":"KM Hermann","year":"2015","unstructured":"Hermann, K.M., Kocisk\u00fd, T., Grefenstette, E., Espeholt, L., Kay, W., Suleyman, M., Blunsom, P.: Teaching machines to read and comprehend. Adv. Neural Inf. Process. Syst. 28, 1693\u20131701 (2015)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"1036_CR12","doi-asserted-by":"crossref","unstructured":"Hochreiter, S., Schmidhuber, J.: Long short-term memory. Neural Comput. (1997)","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"1036_CR13","doi-asserted-by":"crossref","unstructured":"Huang, L., Wang, W., Chen, J., Wei, X.: Attention on attention for image captioning. IEEE, pp. 4633\u20134642 (2019)","DOI":"10.1109\/ICCV.2019.00473"},{"key":"1036_CR14","doi-asserted-by":"crossref","unstructured":"Ji, J., Luo, Y., Sun, X., Chen, F., Luo, G., Wu, Y., Gao, Y., Ji, R.: Improving image captioning by leveraging intra- and inter-layer global representation in transformer network. In: National Conference on Artificial Intelligence (2021)","DOI":"10.1609\/aaai.v35i2.16258"},{"key":"1036_CR15","doi-asserted-by":"crossref","unstructured":"Jiang, H., Misra, I., Rohrbach, M., Learned-Miller, E.G., Chen, X.: In defense of grid features for visual question answering. CoRR abs\/2001.03615 (2020)","DOI":"10.1109\/CVPR42600.2020.01028"},{"key":"1036_CR16","first-page":"510","volume-title":"Recurrent Fusion Network for Image Captioning","author":"W Jiang","year":"2018","unstructured":"Jiang, W., Ma, L., Jiang, Y., Liu, W., Zhang, T.: Recurrent Fusion Network for Image Captioning, pp. 510\u2013526. Springer, Berlin (2018)"},{"key":"1036_CR17","doi-asserted-by":"crossref","unstructured":"Kadlec, R., Schmid, M., Bajgar, O., Kleindienst, J.: Text understanding with the attention sum reader network. In: Proceedings of the 54th Annual Meeting of the Association for Computational Linguistics, Vol. 1: Long Papers (2016)","DOI":"10.18653\/v1\/P16-1086"},{"issue":"4","key":"1036_CR18","doi-asserted-by":"publisher","first-page":"664","DOI":"10.1109\/TPAMI.2016.2598339","volume":"39","author":"A Karpathy","year":"2017","unstructured":"Karpathy, A., Fei-Fei, L.: Deep visual-semantic alignments for generating image descriptions. IEEE Trans. Pattern Anal. Mach. Intell. 39(4), 664\u2013676 (2017)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1036_CR19","doi-asserted-by":"crossref","unstructured":"Lavie, A., Agarwal, A.: Meteor: an automatic metric for MT evaluation with high levels of correlation with human judgments. In: Workshop on Statistical Machine Translation (2007)","DOI":"10.3115\/1626355.1626389"},{"key":"1036_CR20","doi-asserted-by":"crossref","unstructured":"Li, G., Zhu, L., Liu, P., Yang, Y.: Entangled transformer for image captioning. In: 2019 IEEE\/CVF International Conference on Computer Vision (ICCV) (2019)","DOI":"10.1109\/ICCV.2019.00902"},{"key":"1036_CR21","doi-asserted-by":"publisher","first-page":"3260","DOI":"10.3390\/app9163260","volume":"9","author":"J Li","year":"2019","unstructured":"Li, J., Yao, P., Guo, L., Zhang, W.: Boosted transformer for image captioning. Appl. Sci. 9, 3260 (2019). https:\/\/doi.org\/10.3390\/app9163260","journal-title":"Appl. Sci."},{"key":"1036_CR22","unstructured":"Lin, C.Y.: Rouge: a package for automatic evaluation of summaries (2004)"},{"key":"1036_CR23","unstructured":"Liu, F., Liu, Y., Ren, X., He, X., Sun, X.: Aligning visual regions and textual concepts for semantic-grounded image representations. In: 33rd Conference on Neural Information Processing Systems (NeurIPS 2019) (2019)"},{"key":"1036_CR24","unstructured":"Liu, W., Chen, S., Guo, L., Zhu, X., Liu, J.: CPTR: full transformer network for image captioning. CoRR abs\/2101.10804, 2101.10804 (2021)"},{"key":"1036_CR25","doi-asserted-by":"crossref","unstructured":"Lu, J., Xiong, C., Parikh, D., Socher, R.: Knowing when to look: adaptive attention via a visual sentinel for image captioning. IEEE Computer Society, pp. 3242\u20133250 (2017)","DOI":"10.1109\/CVPR.2017.345"},{"key":"1036_CR26","first-page":"2286","volume-title":"Dual-Level Collaborative Transformer for Image Captioning","author":"Y Luo","year":"2021","unstructured":"Luo, Y., Ji, J., Sun, X., Cao, L., Wu, Y., Huang, F., Lin, C., Ji, R.: Dual-Level Collaborative Transformer for Image Captioning, pp. 2286\u20132293. AAAI Press, Palo Alto (2021)"},{"key":"1036_CR27","unstructured":"Messina, N., Falchi, F., Esuli, A., Amato, G.: Transformer reasoning network for image- text matching and retrieval. IEEE, pp. 5222\u20135229 (2020)"},{"key":"1036_CR28","doi-asserted-by":"crossref","unstructured":"Papineni, K., Roukos, S., Ward, T., Zhu, W.: (2002) Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics, July 6\u201312, 2002, Philadelphia, PA, USA","DOI":"10.3115\/1073083.1073135"},{"issue":"6","key":"1036_CR29","doi-asserted-by":"publisher","first-page":"1137","DOI":"10.1109\/TPAMI.2016.2577031","volume":"39","author":"S Ren","year":"2017","unstructured":"Ren, S., He, K., Girshick, R.B., Sun, J.: Faster R-CNN: towards real-time object detection with region proposal networks. IEEE Trans. Pattern Anal. Mach. Intell. 39(6), 1137\u20131149 (2017)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1036_CR30","doi-asserted-by":"crossref","unstructured":"Rennie, S.J., Marcheret, E., Mroueh, Y., Ross, J., Goel, V.: Self-critical sequence training for image captioning. CoRR abs\/1612.00563, 1612.00563 (2016)","DOI":"10.1109\/CVPR.2017.131"},{"key":"1036_CR31","doi-asserted-by":"crossref","unstructured":"Sammani, F., Melas-Kyriazi, L.: Show, edit and tell: a framework for editing image captions. In: 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2020)","DOI":"10.1109\/CVPR42600.2020.00486"},{"key":"1036_CR32","unstructured":"Simonyan, K,, Zisserman, A.: Very deep convolutional networks for large-scale image recognition. In: Bengio, Y., LeCun, Y. (eds) 3rd International Conference on Learning Representations, ICLR 2015, San Diego, CA, USA, May 7\u20139, 2015, Conference Track Proceedings (2015)"},{"key":"1036_CR33","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, L., Polosukhin, I.: Attention is all you need. pp. 5998\u20136008 (2017)"},{"key":"1036_CR34","doi-asserted-by":"crossref","unstructured":"Vedantam, R., Zitnick, CL., Parikh, D.: CIDEr: consensus-based image description evaluation. CoRR abs\/1411.5726 (2014)","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"1036_CR35","unstructured":"Veit, A., Matera, T., Neumann, L., Matas, J., Belongie, S.J .: COCO-Text: dataset and benchmark for text detection and recognition in natural images. CoRR abs\/1601.07140 (2016)"},{"key":"1036_CR36","first-page":"8957","volume-title":"Hierarchical Attention Network for Image Captioning","author":"W Wang","year":"2019","unstructured":"Wang, W., Chen, Z., Hu, H.: Hierarchical Attention Network for Image Captioning, pp. 8957\u20138964. AAAI Press, Palo Alto (2019)"},{"key":"1036_CR37","doi-asserted-by":"crossref","unstructured":"Wei, Y., Wu, C., Li, G., Shi, H.: Sequential transformer via an outside-in attention for image captioning. Eng. Appl. Artif. Intell. 108, 104574 (2022)","DOI":"10.1016\/j.engappai.2021.104574"},{"key":"1036_CR38","doi-asserted-by":"publisher","first-page":"129","DOI":"10.1016\/j.neunet.2022.01.011","volume":"148","author":"T Xian","year":"2022","unstructured":"Xian, T., Li, Z., Zhang, C., Ma, H.: Dual global enhanced transformer for image captioning. Neural Netw. 148, 129\u2013141 (2022)","journal-title":"Neural Netw."},{"key":"1036_CR39","unstructured":"Xu, K., Ba, J., Kiros, R., Cho, K., Courville, A.C., Salakhutdinov, R., Zemel, R.S., Bengio, Y.: (2015) Show, attend and tell: neural image caption generation with visual attention. CoRR abs\/1502.03044"},{"key":"1036_CR40","doi-asserted-by":"crossref","unstructured":"Yang, X., Tang, K., Zhang, H., Cai, J.: Auto-encoding scene graphs for image captioning. Computer Vision Foundation\/IEEE, pp. 10685\u201310694 (2019)","DOI":"10.1109\/CVPR.2019.01094"},{"key":"1036_CR41","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"crossref","first-page":"711","DOI":"10.1007\/978-3-030-01264-9_42","volume-title":"Exploring Visual Relationship for Image Captioning","author":"T Yao","year":"2018","unstructured":"Yao, T., Pan, Y., Li, Y., Mei, T.: Exploring Visual Relationship for Image Captioning. Lecture Notes in Computer Science, vol. 11218, pp. 711\u2013727. Springer, Berlin (2018)"},{"key":"1036_CR42","doi-asserted-by":"crossref","unstructured":"You, Q., Jin, H., Wang, Z., Fang, C., Luo, J.: Image captioning with semantic attention. IEEE Computer Society, pp. 4651\u20134659 (2016)","DOI":"10.1109\/CVPR.2016.503"},{"issue":"12","key":"1036_CR43","doi-asserted-by":"publisher","first-page":"4467","DOI":"10.1109\/TCSVT.2019.2947482","volume":"30","author":"J Yu","year":"2020","unstructured":"Yu, J., Li, J., Yu, Z., Huang, Q.: Multimodal transformer with multi-view visual representation for image captioning. IEEE Trans. Circuits Syst. Video Technol. 30(12), 4467\u20134480 (2020)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"1036_CR44","doi-asserted-by":"crossref","unstructured":"Zhang, X., Sun, X., Luo, Y., Ji, J., Zhou, Y., Wu, Y., Huang, F., Ji, R.: RSTNET: captioning with adaptive attention on visual and non-visual words. In: IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2021, virtual, June 19\u201325, 2021, Computer Vision Foundation\/IEEE, pp. 15465\u201315474 (2021)","DOI":"10.1109\/CVPR46437.2021.01521"},{"key":"1036_CR45","doi-asserted-by":"crossref","unstructured":"Zhu, X., Li, L., Liu, J., Peng, H., Niu, X.: Captioning transformer with stacked attention modules. Appl. Sci. 8 (2018)","DOI":"10.3390\/app8050739"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-022-01036-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-022-01036-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-022-01036-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,5,30]],"date-time":"2023-05-30T14:11:28Z","timestamp":1685455888000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-022-01036-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,12,19]]},"references-count":45,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2023,6]]}},"alternative-id":["1036"],"URL":"https:\/\/doi.org\/10.1007\/s00530-022-01036-z","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,12,19]]},"assertion":[{"value":"26 September 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 December 2022","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"19 December 2022","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors certify that there is no conflict of interest with any individual\/organization for the present work.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}