{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,8]],"date-time":"2026-02-08T01:01:08Z","timestamp":1770512468200,"version":"3.49.0"},"reference-count":64,"publisher":"Springer Science and Business Media LLC","issue":"33-34","license":[{"start":{"date-parts":[[2020,6,18]],"date-time":"2020-06-18T00:00:00Z","timestamp":1592438400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2020,6,18]],"date-time":"2020-06-18T00:00:00Z","timestamp":1592438400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"published-print":{"date-parts":[[2020,9]]},"DOI":"10.1007\/s11042-020-09110-2","type":"journal-article","created":{"date-parts":[[2020,6,18]],"date-time":"2020-06-18T23:07:47Z","timestamp":1592521667000},"page":"24225-24239","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":15,"title":["Boosting image caption generation with feature fusion module"],"prefix":"10.1007","volume":"79","author":[{"given":"Pengfei","family":"Xia","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5404-0003","authenticated-orcid":false,"given":"Jingsong","family":"He","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jin","family":"Yin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2020,6,18]]},"reference":[{"key":"9110_CR1","doi-asserted-by":"crossref","unstructured":"Anderson P, He X, Buehler C, Teney D, Johnson M, Gould S, Zhang L (2018) Bottom-up and top-down attention for image captioning and visual question answering. In: CVPR, vol 3, 6","DOI":"10.1109\/CVPR.2018.00636"},{"key":"9110_CR2","doi-asserted-by":"crossref","unstructured":"Aneja J, Deshpande A, Schwing AG (2018) Convolutional image captioning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 5561\u20135570","DOI":"10.1109\/CVPR.2018.00583"},{"key":"9110_CR3","doi-asserted-by":"crossref","unstructured":"Antol S, Agrawal A, Lu J, Mitchell M, Batra D, Lawrence Zitnick C, Parikh D (2015) Vqa: Visual question answering. In: Proceedings of the IEEE international conference on computer vision, pp 2425\u20132433","DOI":"10.1109\/ICCV.2015.279"},{"key":"9110_CR4","unstructured":"Bahdanau D, Cho K, Bengio Y Neural machine translation by jointly learning to align and translate, arXiv:1409.0473"},{"key":"9110_CR5","unstructured":"Bai S, An S A survey on automatic image caption generation, Neurocomputing"},{"key":"9110_CR6","unstructured":"Banerjee S, Lavie A (2005) Meteor: An automatic metric for mt evaluation with improved correlation with human judgments. In: Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization, pp 65\u201372"},{"key":"9110_CR7","doi-asserted-by":"publisher","first-page":"409","DOI":"10.1613\/jair.4900","volume":"55","author":"R Bernardi","year":"2016","unstructured":"Bernardi R, Cakici R, Elliott D, Erdem A, Erdem E, Ikizler-Cinbis N, Keller F, Muscat A, Plank B (2016) Automatic description generation from images: a survey of models, datasets, and evaluation measures. J Artif Intell Res 55:409\u2013442","journal-title":"J Artif Intell Res"},{"key":"9110_CR8","doi-asserted-by":"crossref","unstructured":"Chen C, Mu S, Xiao W, Ye Z, Wu L, Ju Q (2019) Improving image captioning with conditional generative adversarial nets 33:8142\u20138150","DOI":"10.1609\/aaai.v33i01.33018142"},{"key":"9110_CR9","unstructured":"Chen F, Ji R, Ji J, Sun X, Zhang B, Ge X, Wu Y, Huang F, Wang Y (2019) Variational structured semantic inference for diverse image captioning 1929\u20131939"},{"key":"9110_CR10","unstructured":"Chen L, Zhang H, Xiao J, Nie L, Shao J, Liu W, Chua T-S Sca-cnn: Spatial and channel-wise attention in convolutional networks for image captioning, arXiv:1611.05594"},{"key":"9110_CR11","unstructured":"Chen L-C, Papandreou G, Schroff F, Adam H Rethinking atrous convolution for semantic image segmentation, arXiv:1706.05587"},{"key":"9110_CR12","unstructured":"Chen X, Fang H, Lin T, Vedantam R, Gupta S, Doll??r P, Zitnick CL Microsoft coco captions: Data collection and evaluation server, arXiv:1504.00325"},{"key":"9110_CR13","unstructured":"Cho K, Van Merri\u00ebnboer B, Gulcehre C, Bahdanau D, Bougares F, Schwenk H, Bengio Y Learning phrase representations using rnn encoder-decoder for statistical machine translation, arXiv:1406.1078"},{"key":"9110_CR14","unstructured":"Cornia M, Baraldi L, Cucchiara R Show, control and tell: A framework for generating controllable and grounded captions, arXiv:1811.10652"},{"key":"9110_CR15","unstructured":"Dai B, Fidler S, Urtasun R, Lin D Towards diverse and natural image descriptions via a conditional gan, arXiv:1703.06029"},{"key":"9110_CR16","unstructured":"Deng J, Dong W, Socher R, Li L-J, Li K, Fei-Fei L Imagenet: A large-scale hierarchical image database"},{"key":"9110_CR17","doi-asserted-by":"crossref","unstructured":"Farhadi A, Hejrati M, Sadeghi MA, Young P, Rashtchian C, Hockenmaier J, Forsyth D (2010) Every picture tells a story: Generating sentences from images. In: European conference on computer vision. Springer, New York, pp 15\u201329","DOI":"10.1007\/978-3-642-15561-1_2"},{"key":"9110_CR18","doi-asserted-by":"crossref","unstructured":"Gan Z, Gan C, He X, Pu Y, Tran K, Gao J, Carin L, Deng L (2017) Semantic compositional networks for visual captioning. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 5630\u20135639","DOI":"10.1109\/CVPR.2017.127"},{"key":"9110_CR19","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"9110_CR20","doi-asserted-by":"publisher","first-page":"48","DOI":"10.1016\/j.neucom.2018.02.106","volume":"328","author":"X He","year":"2019","unstructured":"He X, Yang Y, Shi B, Bai X (2019) Vd-san: Visual-densely semantic attention network for image caption generation. Neurocomputing 328:48\u201355","journal-title":"Neurocomputing"},{"key":"9110_CR21","doi-asserted-by":"publisher","first-page":"853","DOI":"10.1613\/jair.3994","volume":"47","author":"M Hodosh","year":"2013","unstructured":"Hodosh M, Young P, Hockenmaier J (2013) Framing image description as a ranking task: Data, models and evaluation metrics. J Artif Intell Res 47:853\u2013899","journal-title":"J Artif Intell Res"},{"issue":"6","key":"9110_CR22","doi-asserted-by":"publisher","first-page":"118","DOI":"10.1145\/3295748","volume":"51","author":"MZ Hossain","year":"2019","unstructured":"Hossain MZ, Sohel F, Shiratuddin MF, Laga H (2019) A comprehensive survey of deep learning for image captioning. ACM Comput Surv 51(6):118","journal-title":"ACM Comput Surv"},{"key":"9110_CR23","doi-asserted-by":"crossref","unstructured":"Hu J, Shen L, Sun G (2018) Squeeze-and-excitation networks. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 7132\u20137141","DOI":"10.1109\/CVPR.2018.00745"},{"key":"9110_CR24","doi-asserted-by":"crossref","unstructured":"Huang G, Liu Z, Van Der Maaten L, Weinberger KQ (2017) Densely connected convolutional networks. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 4700\u20134708","DOI":"10.1109\/CVPR.2017.243"},{"key":"9110_CR25","unstructured":"Huang L, Wang W, Xia Y, Chen J (2019) Adaptively aligned image captioning via adaptive attention time 8940\u20138949"},{"key":"9110_CR26","unstructured":"Ioffe S, Szegedy C Batch normalization:, Accelerating deep network training by reducing internal covariate shift, arXiv:1502.03167"},{"key":"9110_CR27","doi-asserted-by":"crossref","unstructured":"Karpathy A, Fei-Fei L (2015) Deep visual-semantic alignments for generating image descriptions. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 3128\u20133137","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"9110_CR28","doi-asserted-by":"crossref","unstructured":"Khademi M, Schulte O (2018) Image caption generation with hierarchical contextual visual spatial attention. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition Workshops, pp 1943\u20131951","DOI":"10.1109\/CVPRW.2018.00260"},{"key":"9110_CR29","doi-asserted-by":"publisher","first-page":"416","DOI":"10.1016\/j.neucom.2017.07.014","volume":"272","author":"P Kinghorn","year":"2018","unstructured":"Kinghorn P, Zhang L, Shao L (2018) A region-based image caption generator with refined descriptions. Neurocomputing 272:416\u2013424","journal-title":"Neurocomputing"},{"key":"9110_CR30","unstructured":"Kiros R, Salakhutdinov R, Zemel R (2014) Multimodal neural language models. In: International Conference on Machine Learning, pp 595\u2013603"},{"key":"9110_CR31","unstructured":"Krizhevsky A, Sutskever I, Hinton GE (2012) Imagenet classification with deep convolutional neural networks. In: Advances in neural information processing systems, pp 1097\u20131105"},{"issue":"12","key":"9110_CR32","doi-asserted-by":"publisher","first-page":"2891","DOI":"10.1109\/TPAMI.2012.162","volume":"35","author":"G Kulkarni","year":"2013","unstructured":"Kulkarni G, Premraj V, Ordonez V, Dhar S, Li S, Choi Y, Berg AC, Berg TL (2013) Babytalk: Understanding and generating simple image descriptions. IEEE Trans Patt Anal Mach Intell 35(12):2891\u20132903","journal-title":"IEEE Trans Patt Anal Mach Intell"},{"issue":"1","key":"9110_CR33","doi-asserted-by":"publisher","first-page":"351","DOI":"10.1162\/tacl_a_00188","volume":"2","author":"P Kuznetsova","year":"2014","unstructured":"Kuznetsova P, Ordonez V, Berg T, Choi Y (2014) Treetalk: Composition and compression of trees for image descriptions. Trans Assoc Comput Linguis 2(1):351\u2013362","journal-title":"Trans Assoc Comput Linguis"},{"key":"9110_CR34","doi-asserted-by":"crossref","unstructured":"Li L, Tang S, Deng L, Zhang Y, Tian Q (2017) Image caption with global-local attention. In: AAAI, pp 4133\u20134139","DOI":"10.1609\/aaai.v31i1.11236"},{"key":"9110_CR35","unstructured":"Li S, Tao Z, Li K, Fu Y Visual to text: Survey of image and video captioning, IEEE Transactions on Emerging Topics in Computational Intelligence 1\u201316"},{"key":"9110_CR36","unstructured":"Li Z, Peng C, Yu G, Zhang X, Deng Y, Sun J Detnet: A backbone network for object detection, arXiv:1804.06215"},{"key":"9110_CR37","unstructured":"Lin C-Y Rouge: A package for automatic evaluation of summaries, Text Summarization Branches Out"},{"key":"9110_CR38","unstructured":"Lin M, Chen Q, Yan S Network in network, arXiv:1312.4400"},{"key":"9110_CR39","doi-asserted-by":"crossref","unstructured":"Lin T-Y, Doll\u00e1r P, Girshick R, He K, Hariharan B, Belongie S (2017) Feature pyramid networks for object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 2117\u20132125","DOI":"10.1109\/CVPR.2017.106"},{"key":"9110_CR40","doi-asserted-by":"crossref","unstructured":"Liu W, Anguelov D, Erhan D, Szegedy C, Reed S, Fu C-Y, Berg AC (2016) Ssd: Single shot multibox detector. In: European conference on computer vision. Springer, New York, pp 21\u201337","DOI":"10.1007\/978-3-319-46448-0_2"},{"key":"9110_CR41","doi-asserted-by":"crossref","unstructured":"Lu J, Xiong C, Parikh D, Socher R (2017) Knowing when to look: Adaptive attention via a visual sentinel for image captioning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), vol 6, p 2","DOI":"10.1109\/CVPR.2017.345"},{"key":"9110_CR42","unstructured":"Mao J, Xu W, Yang Y, Wang J, Yuille AL Explain images with multimodal recurrent neural networks, arXiv:1410.1090"},{"key":"9110_CR43","unstructured":"Mitchell M, Han X, Dodge J, Mensch A, Goyal A, Berg A, Yamaguchi K, Berg T, Stratos K, Daum\u00e9 H III (2012) Midge: Generating image descriptions from computer vision detections. In: Proceedings of the 13th Conference of the European Chapter of the Association for Computational Linguistics, Association for Computational Linguistics, pp 747\u2013756"},{"key":"9110_CR44","unstructured":"Mnih V, Heess N, Graves A, et al. (2014) Recurrent models of visual attention. In: Advances in neural information processing systems, pp 2204\u20132212"},{"key":"9110_CR45","doi-asserted-by":"crossref","unstructured":"Papineni K, Roukos S, Ward T, Zhu W-J (2002) Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the 40th annual meeting on association for computational linguistics, Association for Computational Linguistics, pp 311\u2013318","DOI":"10.3115\/1073083.1073135"},{"key":"9110_CR46","unstructured":"Paszke A, Gross S, Chintala S, Chanan G, Yang E, DeVito Z, Lin Z, Desmaison A, Antiga L, Lerer A Automatic differentiation in pytorch"},{"key":"9110_CR47","doi-asserted-by":"crossref","unstructured":"Rennie SJ, Marcheret E, Mroueh Y, Ross J, Goel V (2017) Self-critical sequence training for image captioning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 7008\u20137024","DOI":"10.1109\/CVPR.2017.131"},{"key":"9110_CR48","unstructured":"Simonyan K, Zisserman A Very deep convolutional networks for large-scale image recognition, arXiv:1409.1556"},{"issue":"1","key":"9110_CR49","doi-asserted-by":"publisher","first-page":"207","DOI":"10.1162\/tacl_a_00177","volume":"2","author":"R Socher","year":"2014","unstructured":"Socher R, Karpathy A, Le QV, Manning CD, Ng AY (2014) Grounded compositional semantics for finding and describing images with sentences. Trans Assoc Comput Linguis 2(1):207\u2013218","journal-title":"Trans Assoc Comput Linguis"},{"key":"9110_CR50","unstructured":"Sutskever I, Vinyals O, Le QV (2014) Sequence to sequence learning with neural networks. In: Advances in neural information processing systems, pp 3104\u20133112"},{"key":"9110_CR51","doi-asserted-by":"publisher","first-page":"86","DOI":"10.1016\/j.neucom.2018.12.026","volume":"333","author":"YH Tan","year":"2019","unstructured":"Tan YH, Chan CS (2019) Phrase-based image caption generator with hierarchical lstm network. Neurocomputing 333:86\u2013100","journal-title":"Neurocomputing"},{"key":"9110_CR52","doi-asserted-by":"crossref","unstructured":"Vedantam R, Lawrence Zitnick C, Parikh D (2015) Cider: Consensus-based image description evaluation. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 4566\u20134575","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"9110_CR53","doi-asserted-by":"crossref","unstructured":"Vinyals O, Toshev A, Bengio S, Erhan D (2015) Show and tell: A neural image caption generator. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 3156\u20133164","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"9110_CR54","doi-asserted-by":"crossref","unstructured":"Wang C, Yang H, Bartz C, Meinel C (2016) Image captioning with deep bidirectional lstms 988\u2013997","DOI":"10.1145\/2964284.2964299"},{"key":"9110_CR55","doi-asserted-by":"crossref","unstructured":"Wang Q, Chan AB (2019) Describing like humans:, On diversity in image captioning 4195\u20134203","DOI":"10.1109\/CVPR.2019.00432"},{"key":"9110_CR56","doi-asserted-by":"crossref","unstructured":"Wu Q, Shen C, Liu L, Dick A, van den Hengel A (2016) What value do explicit high level concepts have in vision to language problems?. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 203\u2013212","DOI":"10.1109\/CVPR.2016.29"},{"key":"9110_CR57","unstructured":"Xu K, Ba J, Kiros R, Cho K, Courville A, Salakhudinov R, Zemel R, Bengio Y (2015) Show, attend and tell: Neural image caption generation with visual attention. In: International conference on machine learning, pp 2048\u20132057"},{"key":"9110_CR58","unstructured":"Yang Z, Yuan Y, Wu Y, Cohen WW, Salakhutdinov RR (2016) Review networks for caption generation. In: Advances in Neural Information Processing Systems, pp 2361\u20132369"},{"key":"9110_CR59","doi-asserted-by":"crossref","unstructured":"Yao T, Pan Y, Li Y, Qiu Z, Mei T (2017) Boosting image captioning with attributes. In: IEEE International Conference on Computer Vision, ICCV, pp 22\u201329","DOI":"10.1109\/ICCV.2017.524"},{"key":"9110_CR60","doi-asserted-by":"crossref","unstructured":"You Q, Jin H, Wang Z, Fang C, Luo J (2016) Image captioning with semantic attention. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 4651\u20134659","DOI":"10.1109\/CVPR.2016.503"},{"key":"9110_CR61","doi-asserted-by":"crossref","unstructured":"Yu H, Wang J, Huang Z, Yang Y, Xu W (2016) Video paragraph captioning using hierarchical recurrent neural networks. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 4584\u20134593","DOI":"10.1109\/CVPR.2016.496"},{"key":"9110_CR62","doi-asserted-by":"publisher","first-page":"476","DOI":"10.1016\/j.neucom.2018.11.004","volume":"329","author":"D Zhao","year":"2019","unstructured":"Zhao D, Chang Z, Guo S (2019) A multimodal fusion approach for image captioning. Neurocomputing 329:476\u2013485","journal-title":"Neurocomputing"},{"key":"9110_CR63","doi-asserted-by":"crossref","unstructured":"Zhao H, Shi J, Qi X, Wang X, Jia J (2017) Pyramid scene parsing network. In: IEEE Conf. on Computer Vision and Pattern Recognition (CVPR), pp 2881\u20132890","DOI":"10.1109\/CVPR.2017.660"},{"key":"9110_CR64","doi-asserted-by":"publisher","first-page":"55","DOI":"10.1016\/j.neucom.2018.08.069","volume":"319","author":"X Zhu","year":"2018","unstructured":"Zhu X, Li L, Liu J, Li Z, Peng H, Niu X (2018) Image captioning with triple-attention and stack parallel lstm. Neurocomputing 319:55\u201365","journal-title":"Neurocomputing"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-020-09110-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-020-09110-2\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-020-09110-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,10,29]],"date-time":"2022-10-29T10:50:27Z","timestamp":1667040627000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-020-09110-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,6,18]]},"references-count":64,"journal-issue":{"issue":"33-34","published-print":{"date-parts":[[2020,9]]}},"alternative-id":["9110"],"URL":"https:\/\/doi.org\/10.1007\/s11042-020-09110-2","relation":{},"ISSN":["1380-7501","1573-7721"],"issn-type":[{"value":"1380-7501","type":"print"},{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020,6,18]]},"assertion":[{"value":"12 June 2019","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 April 2020","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 May 2020","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 June 2020","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Compliance with Ethical Standards"}},{"value":"The authors declare that they have no conflict of interest. This article does not contain any studies with human participants or animals performed by any of the authors.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"<!--Emphasis Type='Bold' removed-->Conflict of interests"}}]}}