{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,4,2]],"date-time":"2025-04-02T04:27:00Z","timestamp":1743568020129},"reference-count":53,"publisher":"Springer Science and Business Media LLC","issue":"8-9","license":[{"start":{"date-parts":[[2024,5,28]],"date-time":"2024-05-28T00:00:00Z","timestamp":1716854400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,5,28]],"date-time":"2024-05-28T00:00:00Z","timestamp":1716854400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"the Science and Technology Project in Xi\u2019an","award":["No. 22GXFW0123"],"award-info":[{"award-number":["No. 22GXFW0123"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["SIViP"],"published-print":{"date-parts":[[2024,9]]},"DOI":"10.1007\/s11760-024-03268-0","type":"journal-article","created":{"date-parts":[[2024,5,28]],"date-time":"2024-05-28T18:23:31Z","timestamp":1716920611000},"page":"5743-5761","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Research on image caption generation method based on multi-modal pre-training model and text mixup optimization"],"prefix":"10.1007","volume":"18","author":[{"given":"Jing-Tao","family":"Sun","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xuan","family":"Min","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,5,28]]},"reference":[{"key":"3268_CR1","doi-asserted-by":"crossref","unstructured":"Anderson, P., He, X., Buehler, C., et al.: Bottom-up and top-down attention for image captioning and visual question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp.  6077\u20136086 (2018)","DOI":"10.1109\/CVPR.2018.00636"},{"key":"3268_CR2","unstructured":"Gao, L.L., Li, X.P., Song, J.K., Shen, H.T.: Hierarchical LSTMs with adaptive attention for visual captioning. IEEE Trans. Pattern Anal. Mach. Intell. 42(5), 1112\u20131131 (2020)."},{"key":"3268_CR3","doi-asserted-by":"publisher","unstructured":"Zhang, X., Sun, X., Luo, Y., et al.: RSTNet: Captioning with adaptive attention on visual and non-visual words. In: 2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). IEEE, New York (2021). https:\/\/doi.org\/10.1109\/CVPR46437.2021.01521.","DOI":"10.1109\/CVPR46437.2021.01521"},{"key":"3268_CR4","doi-asserted-by":"publisher","first-page":"1271","DOI":"10.1109\/TIP.2019.2940693","volume":"29","author":"H Cui","year":"2019","unstructured":"Cui, H., Zhu, L., Li, J.J., Yang, Y., Nie, L.Q.: Scalable deep hashing for large scale social image retrieval. IEEE Trans. Image Process. 29, 1271\u20131284 (2019)","journal-title":"IEEE Trans. Image Process."},{"key":"3268_CR5","doi-asserted-by":"publisher","unstructured":"Wu, S., Wieland, J., Farivar, O., et al.: Automatic Alt-text: computer-generated image descriptions for blind users on a social network service. In: The 2017 ACM Conference. ACM, New York (2017). https:\/\/doi.org\/10.1145\/2998181.2998364.","DOI":"10.1145\/2998181.2998364"},{"key":"3268_CR6","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2018.2828437","author":"A Das","year":"2016","unstructured":"Das, A., Kottur, S., Gupta, K., et al.: Visual Dialog. (2016). https:\/\/doi.org\/10.1109\/tpami.2018.2828437","journal-title":"Visual Dialog."},{"key":"3268_CR7","doi-asserted-by":"publisher","unstructured":"Jain, U., Lazebnik, S., Schwing, A.: Two can play this game: visual dialog with discriminative question generation and answering. In: 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition. IEEE, New York (2018). https:\/\/doi.org\/10.1109\/CVPR.2018.00603.","DOI":"10.1109\/CVPR.2018.00603"},{"issue":"3","key":"3268_CR8","first-page":"2286","volume":"35","author":"Y Luo","year":"2021","unstructured":"Luo, Y., Ji, J., Sun, X., et al.: Dual-level collaborative transformer for image captioning. Proc AAAI Conf. Artif. Intell. 35(3), 2286\u20132293 (2021)","journal-title":"Proc AAAI Conf. Artif. Intell."},{"key":"3268_CR9","unstructured":"Radford, A., Kim, J. W., Hallacy, C., et al.: Learning transferable visual models from natural language supervision. In: International conference on machine learning. PMLR, pp 8748\u20138763 (2021)."},{"key":"3268_CR10","unstructured":"Li, J., Li, D., Xiong, C., et al.: Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation. In: International conference on machine learning. PMLR, pp 12888\u201312900 (2022)."},{"key":"3268_CR11","doi-asserted-by":"crossref","unstructured":"Fang, Z.Y., Wang, J.F., Hu, X.W., Liang, L., Gan, Z., Wang, L.J., Yang, Y.Z., Liu, Z.C.: Injecting semantic concepts into Endend image captioning. In: Proceedings of the 2022 IEEE\/CVF conference on computer vision and pattern recognition. IEEE, New Orleans, pp 17988\u201317998 (2022).","DOI":"10.1109\/CVPR52688.2022.01748"},{"key":"3268_CR12","doi-asserted-by":"crossref","unstructured":"Wang, Y. Y., Xu, J.G., Sun, Y.F.: Endend transformer based model for image captioning. In: Proceedings of the 36th AAAI conference on artificial intelligence, pp 2585\u20132594. AAAI Press, Palo Alto (2022).","DOI":"10.1609\/aaai.v36i3.20160"},{"key":"3268_CR13","unstructured":"Xu, K., Bajl, Kiros R., et al.: Show, attend and tell: Neural Image caption generation with visual attention. In: 32nd International conference on machine learning. International Machine Learning Society (IMLS), Lile, pp 2048\u20132057 (2015)"},{"key":"3268_CR14","doi-asserted-by":"crossref","unstructured":"Chen, L,, Zhang HW, Xiao J, Nie LQ, Shao J, Liu W, Chua TS. SCACNN: Spatial and channel wise attention in convolutional networks for image captioning. In: Proceedings of the 2017 IEEE Conference on Computer Vision and Pattern Recognition, pp 6298\u20136306. IEEE, Honolulu (2017).","DOI":"10.1109\/CVPR.2017.667"},{"key":"3268_CR15","doi-asserted-by":"crossref","unstructured":"Huang, L., Wang, W.M., Chen, J., Wei, X.Y.: Attention on attention for image captioning. In: Proceedings of the 2019 IEEE\/CVF International Conference on Computer Vision, pp 4633\u20134642. IEEE, Seoul (2019).","DOI":"10.1109\/ICCV.2019.00473"},{"key":"3268_CR16","unstructured":"Liu, W., Chen, S., Guo, L., et al.: CPTR: full transformer network for image captioning. 2021."},{"key":"3268_CR17","unstructured":"Mao, J.H., Xu, W., Yang, Y., Wang, J., Huang, Z.H., Yuille, A.L.: Deep captioning with multi-modal recurrent neural networks (m-RNN). In: Proceedings of the 3rd International conference on learning representations. San Diego, pp 1\u201317 (2015)."},{"key":"3268_CR18","unstructured":"Bao, H., Dong, L., Piao, S., Wei, F.:  BEiT: BERT pre-training of image transformers. In: International Conference on Learning Representations (2021)"},{"key":"3268_CR19","unstructured":"Dong, L., Yang, N., Wang, W., et al.: Unified language model pre-training for natural language understanding and generation. In: Advances in Neural Information Processing Systems 32, Volume 17 of 20: 32nd Conference on Neural Information Processing Systems (NeurIPS 2019), 8\u201314 December 2019, pp. 13019\u201313031. Curran Associates, Inc., Vancouver (2020)"},{"issue":"8","key":"3268_CR20","first-page":"9","volume":"1","author":"A Radford","year":"2019","unstructured":"Radford, A., Wu, J., Child, R., et al.: Language models are unsupervised multitask learners. OpenAI blog 1(8), 9 (2019)","journal-title":"OpenAI blog"},{"key":"3268_CR21","first-page":"1877","volume":"33","author":"T Brown","year":"2020","unstructured":"Brown, T., Mann, B., Ryder, N., et al.: Language models are few-shot learners. Adv. Neural. Inf. Process. Syst. 33, 1877\u20131901 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"3268_CR22","unstructured":"Devlin, J., Chang, M. W., Lee, K, et al.: Bert: Pre-training of deep bidirectional transformers for language understanding. arxiv preprint arXiv:1810.04805, 2018."},{"key":"3268_CR23","unstructured":"Kim, W., Son, B., Kim, I.: Vilt: Vision-and-language transformer without convolution or region supervision. In: International Conference on Machine Learning, pp 5583\u20135594. PMLR,  (2021)."},{"key":"3268_CR24","doi-asserted-by":"publisher","unstructured":"Li, J., Li, D., Xiong, C., et al.: BLIP: Bootstrapping language-image pre-training for unified vision-language understanding and generation (2022). https:\/\/doi.org\/10.48550\/arXiv.2201.12086.","DOI":"10.48550\/arXiv.2201.12086"},{"key":"3268_CR25","unstructured":"Lu, J., Batra, D., Parikh, D., et al.: Vilbert: Pretraining task-agnostic visio linguistic representations for vision-and-language tasks. Adv. Neural Inform. Process. Syst. 32 (2019)."},{"key":"3268_CR26","doi-asserted-by":"crossref","unstructured":"Hu, R., Singh, A.: Unit: Multi-modal multitask learning with a unified transformer. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 1439\u20131449 (2021).","DOI":"10.1109\/ICCV48922.2021.00147"},{"key":"3268_CR27","unstructured":"Guo, H., Mao, Y., Zhang, R.: Augmenting data with mixup for sentence classification: An empirical study. arxiv preprint arXiv:1905.08941 (2019)."},{"key":"3268_CR28","doi-asserted-by":"crossref","unstructured":"Sun, L., Xia, C., Yin, W., et al.: Mixup-transformer: dynamic data augmentation for nlp tasks. arxiv preprint arXiv:2010.02394 (2020).","DOI":"10.18653\/v1\/2020.coling-main.305"},{"key":"3268_CR29","doi-asserted-by":"crossref","unstructured":"Chen, J., Yang, Z., Yang, D.: Mixtext: Linguistically-informed interpolation of hidden space for semi-supervised text classification. arxiv preprint arXiv:2004.12239 (2020).","DOI":"10.18653\/v1\/2020.acl-main.194"},{"key":"3268_CR30","doi-asserted-by":"crossref","unstructured":"Yoon, S., Kim, G., Park, K.: Ssmix: Saliency-based span mixup for text classification. arxiv preprint arXiv:2106.08062 (2021).","DOI":"10.18653\/v1\/2021.findings-acl.285"},{"key":"3268_CR31","unstructured":"Yang, Z., Dai, Z., Yang, Y., et al.: Xlnet: Generalized autoregressive pretraining for language understanding. Adv. Neural Inform. Process. Syst, pp 5754\u20135764 (2019)"},{"key":"3268_CR32","doi-asserted-by":"publisher","unstructured":"Simonyan, K., Zisserman, A.: Very deep convolutional networks for large-scale image recognition. Comp. Sci. (2014). https:\/\/doi.org\/10.48550\/arXiv.1409.1556.","DOI":"10.48550\/arXiv.1409.1556"},{"key":"3268_CR33","unstructured":"Zhang, H., Cisse, M., Dauphin, Y. N., et al.: mixup: Beyond empirical risk minimization. In: International Conference on Learning Representations (2018)"},{"key":"3268_CR34","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.131","author":"SJ Rennie","year":"2016","unstructured":"Rennie, S.J., Marcheret, E., Mroueh, Y., et al.: Self-critical Sequence Training for Image Captioning. IEEE (2016). https:\/\/doi.org\/10.1109\/CVPR.2017.131","journal-title":"IEEE"},{"key":"3268_CR35","doi-asserted-by":"crossref","unstructured":"Vinyals, O., Toshev, A., Bengio, S. et al.: Show and tell: A neural image caption generator. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 3156\u20133164 (2015).","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"3268_CR36","unstructured":"Xu, K.,, Ba.  J., Kiros, R., et al.: Show, attend and tell: Neural image caption generation with visual attention. In: International Conference on Machine Learning, pp 2048\u20132057. PMLR (2015)."},{"key":"3268_CR37","unstructured":"Shen, S., Li, L. H., Tan, H., et al.: How much can CLIP benefit vision-and-language tasks? In: International Conference on Learning Representations (2021)"},{"key":"3268_CR38","doi-asserted-by":"crossref","unstructured":"Zhang, P., Li, X., Hu, X., et al.: Vinvl: revisiting visual representations in vision-language\nmodels. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, \npp. 5579\u20135588 (2021)","DOI":"10.1109\/CVPR46437.2021.00553"},{"key":"3268_CR39","doi-asserted-by":"publisher","unstructured":"Chen, L., Zhang, H., Xiao, J., et al.: SCA-CNN: Spatial and channel-wise attention in convolutional networks for image captioning. In: 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR). IEEE, New York (2016). https:\/\/doi.org\/10.1109\/CVPR.2017.667.","DOI":"10.1109\/CVPR.2017.667"},{"key":"3268_CR40","doi-asserted-by":"publisher","unstructured":"Gan, Z., Gan, C., He, X., et al.: Semantic compositional networks for visual captioning. In: 2017 IEEE conference on computer vision and pattern recognition (CVPR). IEEE, New York (2017). https:\/\/doi.org\/10.1109\/CVPR.2017.127.","DOI":"10.1109\/CVPR.2017.127"},{"key":"3268_CR41","doi-asserted-by":"crossref","unstructured":"Xiaofeng, M. A., Zhao, R., Shi, Z.: Multiscale methods for optical remote-sensing image captioning.\u00a0IEEE Geosci. Remote Sens. Lett. 18(11), 2001\u20132005 (2020).","DOI":"10.1109\/LGRS.2020.3009243"},{"key":"3268_CR42","doi-asserted-by":"publisher","unstructured":"Zhang, Z., Diao, W., Zhang, W., et al.: LAM: Remote sensing image captioning with label-attention mechanism. Remote Sensing 11(20):2349 (2019).https:\/\/doi.org\/10.3390\/rs11202349.","DOI":"10.3390\/rs11202349"},{"key":"3268_CR43","unstructured":"Wang, Z., Yu, J., Yu, A. W., et al.: SimVLM: simple visual language model pretraining with weak supervision. In: International Conference on Learning Representations (2021)."},{"key":"3268_CR44","doi-asserted-by":"crossref","unstructured":"Hu, X. et al.: Scaling up vision-language pre-training for image captioning. In:\u00a0Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 17980\u201317989 (2022).","DOI":"10.1109\/CVPR52688.2022.01745"},{"key":"3268_CR45","doi-asserted-by":"crossref","unstructured":"Yan, K., Ji, L., Luo, H., et al.: Control image captioning spatially and temporally. In: Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers), pp 2014\u20132025 (2021).","DOI":"10.18653\/v1\/2021.acl-long.157"},{"key":"3268_CR46","doi-asserted-by":"crossref","unstructured":"Lin, T. Y., Maire, M., Belongie, S., et al.: Microsoft COCO: Common objects in context. In: 13th European Conference on Computer Vision (ECCV2014), pp 740\u2013755. Springer Verlag, Zurich, Switzerland (2014)","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"3268_CR47","doi-asserted-by":"publisher","DOI":"10.1145\/860435.860459","author":"J Jeon","year":"2003","unstructured":"Jeon, J., Lavrenko, V., Manmatha, R.: Automatic image annotation and retrieval using cross-media relevance models. ACM (2003). https:\/\/doi.org\/10.1145\/860435.860459","journal-title":"ACM"},{"key":"3268_CR48","doi-asserted-by":"crossref","unstructured":"Qu, B.: Deep semantic understanding of high resolution remote sensing image. In: International conference on computer, information and telecommunication systems (Cits), pp 1\u20135. IEEE,  New York (2016)","DOI":"10.1109\/CITS.2016.7546397"},{"key":"3268_CR49","doi-asserted-by":"crossref","unstructured":"Lu, X. et al.: Exploring models and data for remote sensing image caption generation.\u00a0IEEE Trans. Geosci. Remote Sens. 56(4), 2183\u20132195 (2017).","DOI":"10.1109\/TGRS.2017.2776321"},{"key":"3268_CR50","doi-asserted-by":"crossref","unstructured":"Papineni, K. et al.: Bleu: a method for automatic evaluation of machine translation. In:\u00a0Proceedings of the 40th annual meeting of the Association for Computational Linguistics, pp 311\u2013318 (2002).","DOI":"10.3115\/1073083.1073135"},{"key":"3268_CR51","unstructured":"Satanjeev B. M.: An Automatic metric for MT evaluation with improved correlation with human judgments. ACL-2005, 228\u2013231 (2005)."},{"key":"3268_CR52","unstructured":"Lyn, C.: Automatic evaluation of summaries using N-gram cooccurence statistics. In: Proceedings of Human Language Technology Conference (2003)."},{"key":"3268_CR53","doi-asserted-by":"publisher","unstructured":"Vedantam, R., Zitnick, C. L., Parikh, D.: CIDEr: Consensus-based Image Description Evaluation. In: 2015 IEEE Conference on Computer Vision and Pattern Recognition (CVPR). IEEE, New York  (2015).https:\/\/doi.org\/10.1109\/CVPR.2015.7299087.","DOI":"10.1109\/CVPR.2015.7299087"}],"container-title":["Signal, Image and Video Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-024-03268-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11760-024-03268-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-024-03268-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,29]],"date-time":"2024-07-29T19:06:38Z","timestamp":1722279998000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11760-024-03268-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,5,28]]},"references-count":53,"journal-issue":{"issue":"8-9","published-print":{"date-parts":[[2024,9]]}},"alternative-id":["3268"],"URL":"https:\/\/doi.org\/10.1007\/s11760-024-03268-0","relation":{},"ISSN":["1863-1703","1863-1711"],"issn-type":[{"type":"print","value":"1863-1703"},{"type":"electronic","value":"1863-1711"}],"subject":[],"published":{"date-parts":[[2024,5,28]]},"assertion":[{"value":"1 March 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"30 April 2024","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 May 2024","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 May 2024","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interests"}}]}}