{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,3]],"date-time":"2026-02-03T21:02:35Z","timestamp":1770152555603,"version":"3.49.0"},"reference-count":40,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2022,6,14]],"date-time":"2022-06-14T00:00:00Z","timestamp":1655164800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,6,14]],"date-time":"2022-06-14T00:00:00Z","timestamp":1655164800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"published-print":{"date-parts":[[2023,1]]},"DOI":"10.1007\/s11042-022-13279-z","type":"journal-article","created":{"date-parts":[[2022,6,14]],"date-time":"2022-06-14T03:39:21Z","timestamp":1655177961000},"page":"1223-1236","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":16,"title":["A cooperative approach based on self-attention with interactive attribute for image caption"],"prefix":"10.1007","volume":"82","author":[{"given":"Dexin","family":"Zhao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4005-4295","authenticated-orcid":false,"given":"Ruixue","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhaohui","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhiyang","family":"Qi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,6,14]]},"reference":[{"key":"13279_CR1","doi-asserted-by":"crossref","unstructured":"Agrawal H, Desai K, Wang Y, Chen X, Jain R, Johnson M, Batra D, Parikh D, Lee S, Anderson P (2019) Nocaps: novel object captioning at scale. In: Proceedings of international conference on computer vision, pp 8947\u20138956","DOI":"10.1109\/ICCV.2019.00904"},{"key":"13279_CR2","doi-asserted-by":"crossref","unstructured":"Anderson P, He X, Buehler C, Teney D, Johnson M, Gould S, Zhang L (2018) Bottom-up and top-down attention for image captioning and visual question answering. In: Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 6077\u20136086","DOI":"10.1109\/CVPR.2018.00636"},{"key":"13279_CR3","unstructured":"Banerjee S, Lavie A (2005) METEOR: an automatic metric for MT evaluation with improved correlation with human judgments. In: Meeting of the association for computational linguistics, pp. 65\u201372"},{"key":"13279_CR4","doi-asserted-by":"crossref","unstructured":"Chen LC, Zhu Y, Papandreou G, Schroff F, Adam H (2018) Encoder-decoder with atrous separable convolution for semantic image segmentation. arXiv preprint arXiv:1802.02611","DOI":"10.1007\/978-3-030-01234-2_49"},{"key":"13279_CR5","doi-asserted-by":"crossref","unstructured":"Chen H, Ding G, Zhao S (2018) Temporal-difference learning with sampling baseline for image captioning. In: Proceedings of 32nd AAAI conference, pp 6706\u20136713","DOI":"10.1609\/aaai.v32i1.12263"},{"key":"13279_CR6","doi-asserted-by":"crossref","unstructured":"Ding G, Chen M, Zhao S, Chen H et al (2018) Neural image caption generation with weighted training and reference. Cognit Comput 11:763\u2013777","DOI":"10.1007\/s12559-018-9581-x"},{"key":"13279_CR7","doi-asserted-by":"crossref","unstructured":"Fu J, Liu J, Tian H, Li Y, Bao Y, Fang Z, Lu H (2019) Dual attention network for scene segmentation. In: Proceedings of 2019 IEEE\/CVF conference on computer vision and pattern recognition, pp 3146\u20133154","DOI":"10.1109\/CVPR.2019.00326"},{"key":"13279_CR8","doi-asserted-by":"crossref","unstructured":"Girdhar R, Carreira J, Doersch C, Zisserman A (2019) Video action transformer network. In: Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 244\u2013253","DOI":"10.1109\/CVPR.2019.00033"},{"key":"13279_CR9","doi-asserted-by":"crossref","unstructured":"Guo L, Liu J, Yao P (2020) MSCap: multi-style image captioning with unpaired stylized text. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 4204\u20134213","DOI":"10.1109\/CVPR.2019.00433"},{"key":"13279_CR10","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: IEEE Conference on Computer Vision and Pattern Recognition, pp 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"13279_CR11","doi-asserted-by":"crossref","unstructured":"Hu H, Gu J, Zhang Z, Dai J, Wei Y (2018) Relation networks for object detection. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 3588\u20133597","DOI":"10.1109\/CVPR.2018.00378"},{"key":"13279_CR12","unstructured":"Jiasen L, Caiming X, Devi P, Richard S (2017) Knowing when to look: adaptive attention via a visual sentinel for image captioning. In: Computer Science IEEE Conference on Computer Vision and Pattern Recognition, pp 3242\u20133250"},{"key":"13279_CR13","unstructured":"Jiasen L, Jianwei Y, Dhruv B, Devi P (2018) Neural baby talk. In: IEEE Conference on Computer Vision and Pattern Recognition, pp 7219\u20137228"},{"key":"13279_CR14","doi-asserted-by":"crossref","unstructured":"Karpathy, Andrej, Feifei L 2017) Deep visual-semantic alignments for generating image descriptions. In: IEEE Transactions on Pattern Analysis and Machine Intelligence, pp 3128\u20133137","DOI":"10.1109\/TPAMI.2016.2598339"},{"key":"13279_CR15","unstructured":"Kiros R, Salakhutdinov R, Zemel RS et al (2014) Unifying visual-semantic embeddings with multimodal neural language models. arXiv preprint arXiv:1411.2539"},{"issue":"1","key":"13279_CR16","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1007\/s11263-016-0981-7","volume":"123","author":"R Krishna","year":"2017","unstructured":"Krishna R, Zhu Y, Groth O, Johnson J, Hata K, Kravitz J, Chen S, Kalantidis Y, Li LJ, Shamma DA, Bernstein MS, Fei-Fei L (2017) Visual genome: connecting language and vision using crowdsourced dense image annotations. Int J Comput Vis 123(1):32\u201373","journal-title":"Int J Comput Vis"},{"key":"13279_CR17","unstructured":"Lin C (2004) ROUGE: a package for automatic evaluation of summaries. In: Proceedings of Meeting of the association for computational linguistics, pp 74\u201381"},{"key":"13279_CR18","doi-asserted-by":"crossref","unstructured":"Lin T, Maire M, Belongie S, Hays J, Perona P, Ramanan D, Dollar P, Zit-nick (2014) Microsoft coco: common objects in context. In: Proceedings of the European Conference on Computer Vision, pp 740\u2013755","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"13279_CR19","doi-asserted-by":"crossref","unstructured":"Liu W, Anguelov D, Erhan D, Szegedy C, Reed SR, Fu C, Berg AC (2016) Ssd:single shot multibox detector. In: Proceedings of the European Conference on Computer Vision, pp 21\u201337","DOI":"10.1007\/978-3-319-46448-0_2"},{"key":"13279_CR20","doi-asserted-by":"crossref","unstructured":"Lu D, Whitehead S, Huang L, Ji H, Chang S (2018) Entity-aware image caption generation. arXiv preprint arXiv:1804.07889","DOI":"10.18653\/v1\/D18-1435"},{"key":"13279_CR21","unstructured":"Mao J, Xu W, Yang Y, Wang J, Huang Z, Yuille A (2015) Deep captioning with multimodal recurrent neural networks (m-rnn). In: Proceedings of the international conference on learning representations, pp 1\u201317"},{"key":"13279_CR22","doi-asserted-by":"crossref","unstructured":"Papineni K, Roukos S, Ward T et al (2002) Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the meeting of the Association for Computational Linguistics, pp 311\u2013318","DOI":"10.3115\/1073083.1073135"},{"key":"13279_CR23","doi-asserted-by":"crossref","unstructured":"Qin Y, Du J, Zhang Y, Lu H (2019) Look Back and Predict Forward in Image Captioning. In: Proceedings of 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 8359\u20138367","DOI":"10.1109\/CVPR.2019.00856"},{"key":"13279_CR24","doi-asserted-by":"publisher","first-page":"1137","DOI":"10.1109\/TPAMI.2016.2577031","volume":"39","author":"S Ren","year":"2015","unstructured":"Ren S, He K, Girshick R, Sun J (2015) Faster R-CNN: towards real-time object detection with region proposal networks. IEEE Trans Pattern Anal Mach Intell 39:1137\u20131149","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"13279_CR25","doi-asserted-by":"crossref","unstructured":"Rennie S, Marcheret E, Mroueh Y, Ross J, Goel V (2017) Self-critical sequence training for image captioning. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp 7008\u20137024","DOI":"10.1109\/CVPR.2017.131"},{"key":"13279_CR26","doi-asserted-by":"crossref","unstructured":"Sharif N, White L, Bennamoun M et al (2020) WEmbSim: a simple yet effective metric for image captioning. In: 2020 Digital Image Computing: Techniques and Applications (DICTA) (2020):1\u20138","DOI":"10.1109\/DICTA51227.2020.9363392"},{"key":"13279_CR27","unstructured":"Shirai K, Hashimoto K, Eriguchi A et al (2020) Neural text generation with artificial negative examples. ArXiv, abs\/2012.14124"},{"key":"13279_CR28","doi-asserted-by":"crossref","unstructured":"Szegedy C, Liu W, Jia Y, Sermanet P, Reed S, Anguelov D, Erhan D, Vanhoucke V, Rabinovich A (2015) Going deeper with convolutions. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, pp 1\u20139","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"13279_CR29","doi-asserted-by":"crossref","unstructured":"Szegedy C, Vanhoucke V, Ioffe S, Shlens J, Wojn Z (2016) Rethinking the inception architecture for computer vision. In:Proceedings of 2016 IEEE conference on computer vision and pattern recognition, pp 2818\u20132826","DOI":"10.1109\/CVPR.2016.308"},{"key":"13279_CR30","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez A, Kaiser L, Polosukhin I (2017) Attention is all you need. In: Proceedings of Advances in Neural Information Processing Systems neural information processing systems, pp 5998\u20136008"},{"key":"13279_CR31","doi-asserted-by":"crossref","unstructured":"Vedantam R, Zitnick C, Parikh D (2015) CIDEr: consensus-based image description evaluation. In: Proceedings of Computer Vision and Pattern Recognition, pp 4566\u20134575","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"13279_CR32","doi-asserted-by":"crossref","unstructured":"Vinyals O, Toshev A, Bengio S, Erhan D (2015) Show and tell: a neural image caption generator. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, pp 3156\u20133164","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"13279_CR33","doi-asserted-by":"crossref","unstructured":"Wang X, Girshick R, Gupta A, He K (2017) Non-local neural networks. arXiv preprint arXiv:1711.07971, 10","DOI":"10.1109\/CVPR.2018.00813"},{"key":"13279_CR34","unstructured":"Xu K, Ba J, Kiros R et al (2015) Show, attend and tell: neural image caption generation with visual attention. In: Proceedings of International Conference on Machine Learning, pp 2048\u20132057"},{"key":"13279_CR35","doi-asserted-by":"publisher","first-page":"56","DOI":"10.1016\/j.neucom.2018.03.078","volume":"328","author":"J Yang","year":"2019","unstructured":"Yang J, Sun Y, Liang J, Ren B, Lai SH (2019) Image captioning by incorporating affective concepts learned from both visual and textual components. Neurocomputing 328:56\u201368","journal-title":"Neurocomputing"},{"key":"13279_CR36","doi-asserted-by":"crossref","unstructured":"Yang X, Tang K, Zhang H, Cai J (2019) Auto-encoding scene graphs for image captioning. In: Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 10677\u201310686","DOI":"10.1109\/CVPR.2019.01094"},{"key":"13279_CR37","doi-asserted-by":"publisher","unstructured":"Yu J, Li J, Yu Z, Huang Q (2019) Multimodal transformer with multi-view visual representation for image captioning.In: IEEE Trans Circuits Syst Video Technol. https:\/\/doi.org\/10.1109\/TCSVT.2019.2947482","DOI":"10.1109\/TCSVT.2019.2947482"},{"key":"13279_CR38","doi-asserted-by":"publisher","first-page":"476","DOI":"10.1016\/j.neucom.2018.11.004","volume":"329","author":"D Zhao","year":"2019","unstructured":"Zhao D, Chang Z, Guo S (2019) A multimodal fusion approach for image captioning. Neurocomputing 329:476\u2013485","journal-title":"Neurocomputing"},{"key":"13279_CR39","doi-asserted-by":"crossref","unstructured":"Zhao W, Wu X, Zhang X (2020) MemCap: memorizing style knowledge for image captioning. In: Proceedings of the Association for the Advance of Artificial Intelligence, pp 12984\u20132992","DOI":"10.1609\/aaai.v34i07.6998"},{"key":"13279_CR40","doi-asserted-by":"crossref","unstructured":"Zheng Y, Li Y, Wang S (2019) Intention oriented image captions with guiding objects. In: Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 8387\u20138396","DOI":"10.1109\/CVPR.2019.00859"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-022-13279-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-022-13279-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-022-13279-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,12,29]],"date-time":"2022-12-29T01:26:45Z","timestamp":1672277205000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-022-13279-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,6,14]]},"references-count":40,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2023,1]]}},"alternative-id":["13279"],"URL":"https:\/\/doi.org\/10.1007\/s11042-022-13279-z","relation":{},"ISSN":["1380-7501","1573-7721"],"issn-type":[{"value":"1380-7501","type":"print"},{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,6,14]]},"assertion":[{"value":"15 September 2020","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 March 2021","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 May 2022","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 June 2022","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare there is no conflicts of interest regarding the publication of this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}