{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T19:38:27Z","timestamp":1740166707114,"version":"3.37.3"},"reference-count":44,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2022,4,12]],"date-time":"2022-04-12T00:00:00Z","timestamp":1649721600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,4,12]],"date-time":"2022-04-12T00:00:00Z","timestamp":1649721600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key R & D Program of China","doi-asserted-by":"crossref","award":["2019YFC1521204"],"award-info":[{"award-number":["2019YFC1521204"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Multimed Info Retr"],"published-print":{"date-parts":[[2022,6]]},"DOI":"10.1007\/s13735-022-00231-y","type":"journal-article","created":{"date-parts":[[2022,4,12]],"date-time":"2022-04-12T16:55:31Z","timestamp":1649782531000},"page":"149-157","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["A local representation-enhanced recurrent convolutional network for image captioning"],"prefix":"10.1007","volume":"11","author":[{"given":"Xiaoyi","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4939-3880","authenticated-orcid":false,"given":"Jun","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,4,12]]},"reference":[{"key":"231_CR1","doi-asserted-by":"crossref","unstructured":"Anderson P, Fernando B, Johnson M, et al (2016) Spice: semantic propositional image caption evaluation. In: European conference on computer vision. Springer, 382\u2013398","DOI":"10.1007\/978-3-319-46454-1_24"},{"key":"231_CR2","doi-asserted-by":"crossref","unstructured":"Anderson P, He X, Buehler C, et\u00a0al (2018) Bottom-up and top-down attention for image captioning and visual question answering. In: Proceedings of the IEEE conference on computer vision and pattern recognition, 6077\u20136086","DOI":"10.1109\/CVPR.2018.00636"},{"key":"231_CR3","doi-asserted-by":"crossref","unstructured":"Aneja J, Deshpande A, Schwing AG (2018) Convolutional image captioning. In: Proceedings of the IEEE conference on computer vision and pattern recognition, 5561\u20135570","DOI":"10.1109\/CVPR.2018.00583"},{"key":"231_CR4","unstructured":"Banerjee S, Lavie A (2005) Meteor: an automatic metric for mt evaluation with improved correlation with human judgments. In: Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization, 65\u201372"},{"key":"231_CR5","doi-asserted-by":"crossref","unstructured":"Chen L, Zhang H, Xiao J, et al (2017) Sca-cnn: spatial and channel-wise attention in convolutional networks for image captioning. In: Proceedings of the IEEE conference on computer vision and pattern recognition, 5659\u20135667","DOI":"10.1109\/CVPR.2017.667"},{"key":"231_CR6","doi-asserted-by":"crossref","unstructured":"Chen R, Li Z, Zhang D (2019) Adaptive joint attention with reinforcement training for convolutional image caption. In: International workshop on human brain and artificial intelligence. Springer, 235\u2013247","DOI":"10.1007\/978-981-15-1398-5_17"},{"key":"231_CR7","doi-asserted-by":"crossref","unstructured":"Cornia M, Stefanini M, Baraldi L, et al (2020) Meshed-memory transformer for image captioning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, 10,578\u201310,587","DOI":"10.1109\/CVPR42600.2020.01059"},{"issue":"2","key":"231_CR8","doi-asserted-by":"publisher","first-page":"117","DOI":"10.1007\/s13735-018-0151-5","volume":"7","author":"M Dorfer","year":"2018","unstructured":"Dorfer M, Schl\u00fcter J, Vall A et al (2018) End-to-end cross-modality retrieval with cca projections and pairwise ranking loss. Int J Multimed Inf Retrieval 7(2):117\u2013128","journal-title":"Int J Multimed Inf Retrieval"},{"key":"231_CR9","first-page":"1112","volume":"42","author":"L Gao","year":"2019","unstructured":"Gao L, Li X, Song J et al (2019) Hierarchical lstms with adaptive attention for visual captioning. IEEE Trans Pattern Anal Mach Intell 42:1112\u20131131","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"231_CR10","doi-asserted-by":"crossref","unstructured":"Gu J, Wang G, Cai J, et al (2017) An empirical study of language cnn for image captioning. In: Proceedings of the IEEE international conference on computer vision, 1222\u20131231","DOI":"10.1109\/ICCV.2017.138"},{"key":"231_CR11","doi-asserted-by":"crossref","unstructured":"Huang L, Wang W, Chen J, et al (2019) Attention on attention for image captioning. In: Proceedings of the IEEE\/CVF international conference on computer vision, 4634\u20134643","DOI":"10.1109\/ICCV.2019.00473"},{"key":"231_CR12","unstructured":"Huang L, Wang W, Xia Y, et al (2019) Adaptively aligned image captioning via adaptive attention time. arXiv:1909.09060"},{"key":"231_CR13","doi-asserted-by":"publisher","first-page":"7615","DOI":"10.1109\/TIP.2020.3004729","volume":"29","author":"J Ji","year":"2020","unstructured":"Ji J, Xu C, Zhang X et al (2020) Spatio-temporal memory attention for image captioning. IEEE Trans Image Process 29:7615\u20137628","journal-title":"IEEE Trans Image Process"},{"key":"231_CR14","doi-asserted-by":"crossref","unstructured":"Jiang W, Ma L, Jiang YG, et al (2018) Recurrent fusion network for image captioning. In: Proceedings of the European conference on computer vision (ECCV), 499\u2013515","DOI":"10.1007\/978-3-030-01216-8_31"},{"key":"231_CR15","doi-asserted-by":"publisher","first-page":"118","DOI":"10.1016\/j.neucom.2019.08.042","volume":"370","author":"T Jin","year":"2019","unstructured":"Jin T, Li Y, Zhang Z (2019) Recurrent convolutional video captioning with global and local attention. Neurocomputing 370:118\u2013127","journal-title":"Neurocomputing"},{"key":"231_CR16","doi-asserted-by":"crossref","unstructured":"Karpathy A, Fei-Fei L (2015) Deep visual-semantic alignments for generating image descriptions. In: Proceedings of the IEEE conference on computer vision and pattern recognition, 3128\u20133137","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"231_CR17","first-page":"1097","volume":"25","author":"A Krizhevsky","year":"2012","unstructured":"Krizhevsky A, Sutskever I, Hinton GE (2012) Imagenet classification with deep convolutional neural networks. Adv Neural Inf Process Syst 25:1097\u20131105","journal-title":"Adv Neural Inf Process Syst"},{"key":"231_CR18","doi-asserted-by":"crossref","unstructured":"Lan H, Zhang P (2022) Learning and Integrating Multi-Level Matching Features for Image-Text Retrieval. IEEE Signal Process Lett 29:374\u2013378. https:\/\/doi.org\/10.1109\/LSP.2021.3135825","DOI":"10.1109\/LSP.2021.3135825"},{"key":"231_CR19","doi-asserted-by":"publisher","unstructured":"Li G, Zhu L, Liu P, et al (2019) Entangled transformer for image captioning. In: 2019 IEEE\/CVF international conference on computer vision (ICCV), 8927\u20138936. https:\/\/doi.org\/10.1109\/ICCV.2019.00902","DOI":"10.1109\/ICCV.2019.00902"},{"key":"231_CR20","unstructured":"Lin CY (2004) Rouge: a package for automatic evaluation of summaries. In: Text summarization branches out, 74\u201381"},{"key":"231_CR21","doi-asserted-by":"crossref","unstructured":"Lin TY, Maire M, Belongie S, et al (2014) Microsoft coco: common objects in context. In: European conference on computer vision. Springer, 740\u2013755","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"231_CR22","doi-asserted-by":"crossref","unstructured":"Lu J, Xiong C, Parikh D, et al (2017) Knowing when to look: adaptive attention via a visual sentinel for image captioning. In: Proceedings of the IEEE conference on computer vision and pattern recognition, 375\u2013383","DOI":"10.1109\/CVPR.2017.345"},{"key":"231_CR23","doi-asserted-by":"crossref","unstructured":"Pan Y, Yao T, Li Y, et al (2020) X-linear attention networks for image captioning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","DOI":"10.1109\/CVPR42600.2020.01098"},{"key":"231_CR24","doi-asserted-by":"crossref","unstructured":"Papineni K, Roukos S, Ward T, et\u00a0al (2002) Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the 40th annual meeting of the Association for Computational Linguistics, 311\u2013318","DOI":"10.3115\/1073083.1073135"},{"key":"231_CR25","doi-asserted-by":"crossref","unstructured":"Qin Y, Du J, Zhang Y, et al (2019) Look back and predict forward in image captioning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, 8367\u20138375","DOI":"10.1109\/CVPR.2019.00856"},{"key":"231_CR26","first-page":"91","volume":"28","author":"S Ren","year":"2015","unstructured":"Ren S, He K, Girshick R et al (2015) Faster r-cnn: towards real-time object detection with region proposal networks. Adv Neural Inf Process Syst 28:91\u201399","journal-title":"Adv Neural Inf Process Syst"},{"key":"231_CR27","doi-asserted-by":"crossref","unstructured":"Sharma P, Ding N, Goodman S, et al (2018) Conceptual captions: a cleaned, hypernymed, image alt-text dataset for automatic image captioning. In: Proceedings of the 56th annual meeting of the association for computational linguistics (Volume 1: Long Papers), 2556\u20132565","DOI":"10.18653\/v1\/P18-1238"},{"key":"231_CR28","doi-asserted-by":"crossref","unstructured":"Vedantam R, Lawrence\u00a0Zitnick C, Parikh D (2015) Cider: consensus-based image description evaluation. In: Proceedings of the IEEE conference on computer vision and pattern recognition, 4566\u20134575","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"231_CR29","doi-asserted-by":"crossref","unstructured":"Vinyals O, Toshev A, Bengio S, et al (2015) Show and tell: a neural image caption generator. In: Proceedings of the IEEE conference on computer vision and pattern recognition, 3156\u20133164","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"231_CR30","doi-asserted-by":"crossref","unstructured":"Wang C, Gu X (2021) Image captioning with adaptive incremental global context attention. Appl Intell. https:\/\/doi.org\/10.1007\/s10489-021-02734-3","DOI":"10.1007\/s10489-021-02734-3"},{"key":"231_CR31","doi-asserted-by":"crossref","unstructured":"Wang L, Bai Z, Zhang Y, et al (2020) Show, recall, and tell: Image captioning with recall mechanism. In: Proceedings of the AAAI conference on artificial intelligence, 12,176\u201312,183","DOI":"10.1609\/aaai.v34i07.6898"},{"key":"231_CR32","unstructured":"Wang Q, Chan AB (2018) Cnn+ cnn: convolutional decoders for image captioning. arXiv:1805.09019"},{"key":"231_CR33","doi-asserted-by":"crossref","unstructured":"Wang W, Chen Z, Hu H (2019) Hierarchical attention network for image captioning. In: Proceedings of the AAAI conference on artificial intelligence, 8957\u20138964","DOI":"10.1609\/aaai.v33i01.33018957"},{"issue":"107","key":"231_CR34","first-page":"313","volume":"228","author":"Y Wang","year":"2021","unstructured":"Wang Y, Sun X, Li X et al (2021) Reasoning like humans: on dynamic attention prior in image captioning. Knowl-Based Syst 228(107):313","journal-title":"Knowl-Based Syst"},{"key":"231_CR35","doi-asserted-by":"publisher","first-page":"2","DOI":"10.1145\/3439734","volume":"17","author":"H Wei","year":"2021","unstructured":"Wei H, Li Z, Huang F et al (2021) Integrating scene semantic knowledge into image captioning. ACM Trans Multimed Comput Commun Appl 17:2. https:\/\/doi.org\/10.1145\/3439734","journal-title":"ACM Trans Multimed Comput Commun Appl"},{"issue":"11","key":"231_CR36","doi-asserted-by":"publisher","first-page":"4299","DOI":"10.1109\/TCSVT.2019.2956593","volume":"30","author":"A Wu","year":"2019","unstructured":"Wu A, Han Y, Yang Y et al (2019) Convolutional reconstruction-to-sequence for video captioning. IEEE Trans Circuits Syst Video Technol 30(11):4299\u20134308","journal-title":"IEEE Trans Circuits Syst Video Technol"},{"issue":"4","key":"231_CR37","first-page":"1","volume":"14","author":"J Wu","year":"2018","unstructured":"Wu J, Hu H, Wu Y (2018) Image captioning via semantic guidance attention and consensus selection strategy. ACM Trans Multimed Comput Commun Appl (TOMM) 14(4):1\u201319","journal-title":"ACM Trans Multimed Comput Commun Appl (TOMM)"},{"key":"231_CR38","doi-asserted-by":"publisher","first-page":"173","DOI":"10.1016\/j.patrec.2019.11.003","volume":"129","author":"H Xiao","year":"2020","unstructured":"Xiao H, Xu J, Shi J (2020) Exploring diverse and fine-grained caption for video by incorporating convolutional architecture into lstm-based model. Pattern Recognit Lett 129:173\u2013180","journal-title":"Pattern Recognit Lett"},{"key":"231_CR39","unstructured":"Xu K, Ba J, Kiros R, et al (2015) Show, attend and tell: neural image caption generation with visual attention. In: International conference on machine learning, PMLR, 2048\u20132057"},{"key":"231_CR40","doi-asserted-by":"crossref","unstructured":"Yang L, Tang K, Yang J, et\u00a0al (2017) Dense captioning with joint inference and visual context. In: Proceedings of the IEEE conference on computer vision and pattern recognition, 2193\u20132202","DOI":"10.1109\/CVPR.2017.214"},{"key":"231_CR41","doi-asserted-by":"publisher","first-page":"835","DOI":"10.1109\/TMM.2020.2990074","volume":"23","author":"L Yang","year":"2020","unstructured":"Yang L, Wang H, Tang P et al (2020) Captionnet: a tailor-made recurrent neural network for generating image descriptions. IEEE Trans Multimed 23:835\u2013845","journal-title":"IEEE Trans Multimed"},{"key":"231_CR42","doi-asserted-by":"crossref","unstructured":"Yao T, Pan Y, Li Y, et\u00a0al (2018) Exploring visual relationship for image captioning. In: Proceedings of the European conference on computer vision (ECCV), 684\u2013699","DOI":"10.1007\/978-3-030-01264-9_42"},{"key":"231_CR43","doi-asserted-by":"crossref","unstructured":"You Q, Jin H, Wang Z, et\u00a0al (2016) Image captioning with semantic attention. In: Proceedings of the IEEE conference on computer vision and pattern recognition, 4651\u20134659","DOI":"10.1109\/CVPR.2016.503"},{"key":"231_CR44","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1162\/tacl_a_00166","volume":"2","author":"P Young","year":"2014","unstructured":"Young P, Lai A, Hodosh M et al (2014) From image descriptions to visual denotations: new similarity metrics for semantic inference over event descriptions. Trans Assoc Comput Linguist 2:67\u201378","journal-title":"Trans Assoc Comput Linguist"}],"container-title":["International Journal of Multimedia Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-022-00231-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13735-022-00231-y\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-022-00231-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,5,6]],"date-time":"2022-05-06T17:03:58Z","timestamp":1651856638000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13735-022-00231-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,4,12]]},"references-count":44,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2022,6]]}},"alternative-id":["231"],"URL":"https:\/\/doi.org\/10.1007\/s13735-022-00231-y","relation":{},"ISSN":["2192-6611","2192-662X"],"issn-type":[{"type":"print","value":"2192-6611"},{"type":"electronic","value":"2192-662X"}],"subject":[],"published":{"date-parts":[[2022,4,12]]},"assertion":[{"value":"28 December 2021","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 March 2022","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 March 2022","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 April 2022","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"This paper contains no cases of studies with human participants performed by any of the authors","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval"}},{"value":"Not applicable.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to participate"}},{"value":"Not applicable.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}},{"value":"Not applicable.","order":6,"name":"Ethics","group":{"name":"EthicsHeading","label":"Code availability"}}]}}