{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,25]],"date-time":"2025-12-25T07:26:30Z","timestamp":1766647590076,"version":"3.28.0"},"reference-count":53,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2024,1,24]],"date-time":"2024-01-24T00:00:00Z","timestamp":1706054400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,24]],"date-time":"2024-01-24T00:00:00Z","timestamp":1706054400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Cogn Comput"],"published-print":{"date-parts":[[2024,5]]},"DOI":"10.1007\/s12559-023-10231-7","type":"journal-article","created":{"date-parts":[[2024,1,24]],"date-time":"2024-01-24T14:14:08Z","timestamp":1706105648000},"page":"1061-1072","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Multi-Keys Attention Network for Image Captioning"],"prefix":"10.1007","volume":"16","author":[{"given":"Ziqian","family":"Yang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hui","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Renrong","family":"Ouyang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Quan","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jimin","family":"Xiao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,1,24]]},"reference":[{"issue":"12","key":"10231_CR1","doi-asserted-by":"publisher","first-page":"2891","DOI":"10.1109\/TPAMI.2012.162","volume":"35","author":"G Kulkarni","year":"2013","unstructured":"Kulkarni G, Premraj V, Ordonez V, Dhar S, Li S, Choi Y, Berg AC, Berg TL. Babytalk: understanding and generating simple image descriptions. IEEE Trans Pattern Anal Mach Intell. 2013;35(12):2891.","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"10231_CR2","doi-asserted-by":"crossref","unstructured":"Fang H, Gupta S, Iandola F, Srivastava RK, Deng L, Doll\u00e1r P, Gao J, He X, Mitchell M, Platt JC, et\u00a0al. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 2015. p. 1473\u201382.","DOI":"10.1109\/CVPR.2015.7298754"},{"key":"10231_CR3","unstructured":"Mitchell M, Dodge J, Goyal A, Yamaguchi K, Stratos K, Han X, Mensch A, Berg A, Berg T, Daum\u00e9\u00a0III, H. In: Proceedings of the 13th Conference of the European Chapter of the Association for Computational Linguistics. 2012. p. 747\u201356."},{"key":"10231_CR4","unstructured":"Li Y, Pan Y, Yao T, Mei T. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 2022. p. 17990\u20139."},{"issue":"1","key":"10231_CR5","doi-asserted-by":"publisher","first-page":"539","DOI":"10.1109\/TPAMI.2022.3148210","volume":"45","author":"M Stefanini","year":"2022","unstructured":"Stefanini M, Cornia M, Baraldi L, Cascianelli S, Fiameni G, Cucchiara R. From show to tell: a survey on deep learning-based image captioning. IEEE Trans Pattern Anal Mach Intell. 2022;45(1):539.","journal-title":"IEEE Trans Pattern Anal Mach Intell."},{"key":"10231_CR6","unstructured":"Sutskever I, Vinyals O, Le QV. Sequence to sequence learning with neural networks. Adv Neural Inf Proces Syst. 2014;27."},{"key":"10231_CR7","unstructured":"Xu K, Ba J, Kiros R, Cho K, Courville A, Salakhudinov R, Zemel R, Bengio Y. In: International Conference on Machine Learning. PMLR; 2015. p. 2048\u201357."},{"key":"10231_CR8","doi-asserted-by":"crossref","unstructured":"Anderson P, He X, Buehler C, Teney D, Johnson M, Gould S, Zhang L. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 2018. p. 6077\u201386.","DOI":"10.1109\/CVPR.2018.00636"},{"key":"10231_CR9","doi-asserted-by":"publisher","first-page":"807","DOI":"10.1007\/s12559-019-09656-w","volume":"13","author":"H Chen","year":"2021","unstructured":"Chen H, Ding G, Lin Z, Guo Y, Shan C, Han J. Image captioning with memorized knowledge. Cognitive Computation. 2021;13:807.","journal-title":"Cognitive Computation."},{"key":"10231_CR10","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser \u0141, Polosukhin I. Attention is all you need. Adv Neural Inf Proces Syst. 2017;30."},{"key":"10231_CR11","unstructured":"Y.N. Dauphin, A.\u00a0Fan, M.\u00a0Auli, D.\u00a0Grangier. In: International Conference on Machine Learning. PMLR; 2017. p. 933\u201341."},{"key":"10231_CR12","unstructured":"You Q, Jin H, Wang Z, Fang C, Luo J. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 2016. p. 4651\u20139."},{"key":"10231_CR13","unstructured":"Lu J, Xiong C, Parikh D, Socher R. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 2017. p. 375\u201383."},{"key":"10231_CR14","unstructured":"Huang L, Wang W, Chen J, Wei XY. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. 2019. p. 4634\u201343."},{"key":"10231_CR15","unstructured":"Pan Y, Yao T, Li Y, Mei T. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 2020. p. 10971\u201380."},{"key":"10231_CR16","unstructured":"Zhou Y, Hu Z, Liu D, Ben H, Wang M. Compact bidirectional transformer for image captioning. arXiv:2201.01984 [Preprint]. 2022."},{"key":"10231_CR17","doi-asserted-by":"crossref","unstructured":"Farhadi A, Hejrati M, Sadeghi MA, Young P, Rashtchian C, Hockenmaier J, Forsyth D. In: Computer Vision\u2013ECCV 2010: 11th European Conference on Computer Vision, Heraklion, Crete, Greece, September 5-11, 2010, Proceedings, Part IV 11. Springer, 2010. p. 15\u201329.","DOI":"10.1007\/978-3-642-15561-1_2"},{"key":"10231_CR18","unstructured":"Li S, Kulkarni G, Berg T, Berg A, Choi Y. In: Proceedings of the Fifteenth Conference on Computational Natural Language Learning. 2011. p. 220\u20138."},{"key":"10231_CR19","doi-asserted-by":"crossref","unstructured":"Gong Y, Wang L, Hodosh M, Hockenmaier J, Lazebnik S. In: Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6\u201312, 2014, Proceedings, Part IV 13. Springer; 2014. p. 529\u201345.","DOI":"10.1007\/978-3-319-10593-2_35"},{"key":"10231_CR20","doi-asserted-by":"publisher","first-page":"853","DOI":"10.1613\/jair.3994","volume":"47","author":"M Hodosh","year":"2013","unstructured":"Hodosh M, Young P, Hockenmaier J. Framing image description as a ranking task: data, models and evaluation metrics. J Artif Intell Res. 2013;47:853.","journal-title":"J Artif Intell Res"},{"key":"10231_CR21","unstructured":"Ordonez V, Kulkarni G, Berg T. Im2text: describing images using 1 million captioned photographs. Adv Neural Inf Proces Syst. 2011;24."},{"key":"10231_CR22","unstructured":"Sun C, Gan C, Nevatia R. In: Proceedings of the IEEE International Conference on Computer Vision. 2015. p. 2596\u2013604."},{"key":"10231_CR23","doi-asserted-by":"crossref","unstructured":"Cho K, Van\u00a0Merri\u00ebnboer B, Gulcehre C, Bahdanau D, Bougares F, Schwenk H, Bengio Y. Learning phrase representations using RNN encoder-decoder for statistical machine translation. arXiv:1406.1078 [Preprint]. 2014.","DOI":"10.3115\/v1\/D14-1179"},{"key":"10231_CR24","unstructured":"Mao J, Xu W, Yang Y, Wang J, Yuille AL. Explain images with multimodal recurrent neural networks. arXiv:1410.1090 [Preprint]. 2014."},{"key":"10231_CR25","doi-asserted-by":"crossref","unstructured":"Vinyals O, Toshev A, Bengio S, Erhan D. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 2015. p. 3156\u201364.","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"10231_CR26","unstructured":"Xu D, Zhu Y, Choy CB, Fei-Fei L. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 2017. p. 5410\u20139."},{"key":"10231_CR27","unstructured":"Yang J, Lu J, Lee S, Batra D, Parikh D. In: Proceedings of the European Conference on Computer Vision (ECCV). 2018. p. 670\u201385."},{"key":"10231_CR28","unstructured":"Yang X, Tang K, Zhang H, Cai J. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 2019. p. 10685\u201394."},{"issue":"11","key":"10231_CR29","doi-asserted-by":"publisher","first-page":"2942","DOI":"10.1109\/TMM.2019.2915033","volume":"21","author":"X Xiao","year":"2019","unstructured":"Xiao X, Wang L, Ding K, Xiang S, Pan C. Deep hierarchical encoder-decoder network for image captioning. IEEE Trans Multimedia. 2019;21(11):2942.","journal-title":"IEEE Trans Multimedia"},{"key":"10231_CR30","unstructured":"Shetty R, Rohrbach M, Anne\u00a0Hendricks L, Fritz M, Schiele B. In: Proceedings of the IEEE International Conference on Computer Vision. 2017. p. 4135\u201344."},{"key":"10231_CR31","doi-asserted-by":"crossref","unstructured":"Chen C, Mu S, Xiao W, Ye Z, Wu L, Ju Q. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a033. 2019. p. 8142\u201350.","DOI":"10.1609\/aaai.v33i01.33018142"},{"key":"10231_CR32","doi-asserted-by":"crossref","unstructured":"Vedantam R, Lawrence\u00a0Zitnick C, Parikh D. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 2015. p. 4566\u201375.","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"10231_CR33","unstructured":"Luo R. A better variant of self-critical sequence training. arXiv:2003.09971 [Preprint]. 2020."},{"key":"10231_CR34","doi-asserted-by":"crossref","unstructured":"Lin TY, Maire M, Belongie S, Hays J, Perona P, Ramanan D, Doll\u00e1r P, Zitnick CL. In: Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6\u201312, 2014, Proceedings, Part V 13. Springer; 2014. p. 740\u201355.","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"10231_CR35","doi-asserted-by":"crossref","unstructured":"Karpathy A, Fei-Fei L. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 2015. p. 3128\u201337.","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"10231_CR36","doi-asserted-by":"crossref","unstructured":"Papineni K, Roukos S, Ward T, Zhu WJ. In: Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics. 2002. p. 311\u20138.","DOI":"10.3115\/1073083.1073135"},{"key":"10231_CR37","unstructured":"Banerjee S, Lavie A. In: Proceedings of the ACL workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization. 2005. p. 65\u201372."},{"key":"10231_CR38","unstructured":"Lin CY. In: Text summarization branches out. 2004. p. 74\u201381."},{"key":"10231_CR39","doi-asserted-by":"crossref","unstructured":"Anderson P, Fernando B, Johnson M, Gould S. In: Computer Vision\u2013ECCV 2016: 14th European Conference, Amsterdam, The Netherlands, October 11\u201314, 2016, Proceedings, Part V 14. Springer; 2016. p. 382\u201398.","DOI":"10.1007\/978-3-319-46454-1_24"},{"key":"10231_CR40","unstructured":"Cornia M, Stefanini M, Baraldi L, Cucchiara R. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 2020. p. 10578\u201387."},{"key":"10231_CR41","doi-asserted-by":"crossref","unstructured":"Yang X, Zhang H, Cai J. Deconfounded image captioning: a causal retrospect. IEEE Trans Pattern Anal Mach Intell. 2021.","DOI":"10.1109\/TPAMI.2021.3121705"},{"key":"10231_CR42","doi-asserted-by":"crossref","unstructured":"Wang J, Xu W, Wang Q, Chan AB. On distinctive image captioning via comparing and reweighting. IEEE Trans Pattern Anal Mach Intell. 2022;45(2):2088.","DOI":"10.1109\/TPAMI.2022.3159811"},{"key":"10231_CR43","unstructured":"Ren S, He K, Girshick R, Sun J. Faster r-CNN: towards real-time object detection with region proposal networks. Adv Neural Inf Proces Syst. 2015;28."},{"key":"10231_CR44","unstructured":"Zhang P, Li X, Hu X, Yang J, Zhang L, Wang L, Choi Y, Gao J. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 2021. p. 5579\u201388."},{"key":"10231_CR45","unstructured":"Yao T, Pan Y, Li Y, Qiu Z, Mei T. In: Proceedings of the IEEE international conference on computer vision. 2017. p. 4894\u2013902."},{"key":"10231_CR46","unstructured":"Rennie SJ, Marcheret E, Mroueh Y, Ross J, Goel V. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 2017. p. 7008\u201324."},{"key":"10231_CR47","unstructured":"Jiang W, Ma L, Jiang YG, Liu W, Zhang T. In: Proceedings of the European Conference on Computer Vision (ECCV). 2018. p. 499\u2013515."},{"key":"10231_CR48","unstructured":"Yao T, Pan Y, Li Y, Mei T. In: Proceedings of the European Conference on Computer Vision (ECCV). 2018. p. 684\u201399."},{"key":"10231_CR49","unstructured":"Qin Y, Du J, Zhang Y, Lu H, In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 2019. p. 8367\u201375."},{"key":"10231_CR50","unstructured":"Zhang X, Sun X, Luo Y, Ji J, Zhou Y, Wu Y, Huang F, Ji R. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 2021. p. 15465\u201374."},{"key":"10231_CR51","unstructured":"Yao T, Pan Y, Li Y, Mei T. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. 2019. p. 2621\u20139."},{"key":"10231_CR52","unstructured":"Li G, Zhu L, Liu P, Yang Y. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. 2019. p. 8928\u201337."},{"key":"10231_CR53","doi-asserted-by":"crossref","unstructured":"Ji J, Luo Y, Sun X, Chen F, Luo G, Wu Y, Gao Y, Ji R. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a035. 2021. p. 1655\u201363.","DOI":"10.1609\/aaai.v35i2.16258"}],"container-title":["Cognitive Computation"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s12559-023-10231-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s12559-023-10231-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s12559-023-10231-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,9]],"date-time":"2024-11-09T01:22:45Z","timestamp":1731115365000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s12559-023-10231-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,1,24]]},"references-count":53,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2024,5]]}},"alternative-id":["10231"],"URL":"https:\/\/doi.org\/10.1007\/s12559-023-10231-7","relation":{},"ISSN":["1866-9956","1866-9964"],"issn-type":[{"type":"print","value":"1866-9956"},{"type":"electronic","value":"1866-9964"}],"subject":[],"published":{"date-parts":[[2024,1,24]]},"assertion":[{"value":"24 February 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 November 2023","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 January 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"This article does not contain any studies with human participants performed by any of the authors.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical Approval"}},{"value":"The authors declare no competing interests.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of Interest"}}]}}