{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T17:24:52Z","timestamp":1777656292661,"version":"3.51.4"},"reference-count":103,"publisher":"Springer Science and Business Media LLC","issue":"9","license":[{"start":{"date-parts":[[2023,8,31]],"date-time":"2023-08-31T00:00:00Z","timestamp":1693440000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,8,31]],"date-time":"2023-08-31T00:00:00Z","timestamp":1693440000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"DOI":"10.1007\/s11042-023-16560-x","type":"journal-article","created":{"date-parts":[[2023,8,31]],"date-time":"2023-08-31T09:02:23Z","timestamp":1693472543000},"page":"28077-28123","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":10,"title":["From methods to datasets: A survey on Image-Caption Generators"],"prefix":"10.1007","volume":"83","author":[{"given":"Lakshita","family":"Agarwal","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3534-3364","authenticated-orcid":false,"given":"Bindu","family":"Verma","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,8,31]]},"reference":[{"key":"16560_CR1","unstructured":"Wikipedia contributors (2022) Photo caption - Wikipedia, The Free Encyclopedia. [Online; accessed 28-February-2022]"},{"key":"16560_CR2","doi-asserted-by":"crossref","unstructured":"Chen, F., Li, X., Tang, J., Li, S., Wang, T.: A survey on recent advances in image captioning. In: Journal of Physics: Conference Series, vol. 1914, p. 012053 (2021). IOP Publishing","DOI":"10.1088\/1742-6596\/1914\/1\/012053"},{"key":"16560_CR3","unstructured":"Elhagry, A., Kadaoui, K.: A thorough review on recent deep learning methodologies for image captioning. arXiv preprint arXiv:2107.13114 (2021)"},{"key":"16560_CR4","unstructured":"Stefanini, M., Cornia, M., Baraldi, L., Cascianelli, S., Fiameni, G., Cucchiara, R.: From show to tell: A survey on image captioning. arXiv preprint arXiv:2107.06912 (2021)"},{"key":"16560_CR5","doi-asserted-by":"crossref","unstructured":"Wang, H., Zhang, Y., Yu, X.: An overview of image caption generation methods. Computational intelligence and neuroscience 2020 (2020)","DOI":"10.1155\/2020\/3062706"},{"key":"16560_CR6","doi-asserted-by":"crossref","unstructured":"Mao, J., Wei, X., Yang, Y., Wang, J., Huang, Z., Yuille, A.L.: Learning like a child: Fast novel visual concept learning from sentence descriptions of images. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2533-2541 (2015)","DOI":"10.1109\/ICCV.2015.291"},{"key":"16560_CR7","unstructured":"by Saheel, S.: Baby talk: Understanding and generating image descriptions"},{"key":"16560_CR8","unstructured":"Ordonez, V., Kulkarni, G., Berg, T.: Im2text: Describing images using 1 million captioned photographs. Advances in neural information processing systems 24 (2011)"},{"key":"16560_CR9","doi-asserted-by":"crossref","unstructured":"Chen, X., Zitnick, C.L.: Learning a recurrent visual representation for image caption generation. arXiv preprint arXiv:1411.5654 (2014)","DOI":"10.1109\/CVPR.2015.7298856"},{"key":"16560_CR10","doi-asserted-by":"crossref","unstructured":"Jeon, J., Lavrenko, V., Manmatha, R.: Automatic image annotation and retrieval using cross-media relevance models. In: Proceedings of the 26th Annual International ACM SIGIR Conference on Research and Development in Informaion Retrieval, pp. 119-126 (2003)","DOI":"10.1145\/860435.860459"},{"issue":"9","key":"16560_CR11","doi-asserted-by":"publisher","first-page":"1075","DOI":"10.1109\/TPAMI.2003.1227984","volume":"25","author":"J Li","year":"2003","unstructured":"Li J, Wang JZ (2003) Automatic linguistic indexing of pictures by a statistical modeling approach. IEEE Transactions on pattern analysis and machine intelligence 25(9):1075\u20131088","journal-title":"IEEE Transactions on pattern analysis and machine intelligence"},{"key":"16560_CR12","unstructured":"H\u00e9de, P., Mo\u00ebllic, P.-A., Bourgeoys, J., Joint, M., Thomas, C.: Automatic generation of natural language description for images. In: RIAO, pp. 306-313 (2004). Citeseer"},{"key":"16560_CR13","unstructured":"Pan J-Y, Yang H-J, Duygulu P, Faloutsos C (2004) Automatic image captioning. In: 2004 IEEE International Conference on Multimedia and Expo (ICME)(IEEE Cat. No. 04TH8763), vol. 3, pp. 1987-1990. IEEE"},{"key":"16560_CR14","unstructured":"Li S, Kulkarni G, Berg T, Berg A, Choi Y (2011) Composing simple image descriptions using web-scale n-grams. In: Proceedings of the Fifteenth Conference on Computational Natural Language Learning, pp. 220-228"},{"key":"16560_CR15","doi-asserted-by":"crossref","unstructured":"Mason R, Charniak E (2014) Domain-specific image captioning. In: Proceedings of the Eighteenth Conference on Computational Natural Language Learning, pp. 11-20","DOI":"10.3115\/v1\/W14-1602"},{"key":"16560_CR16","doi-asserted-by":"crossref","unstructured":"Han S-H, Choi H-J (2020) Domain-specific image caption generator with semantic ontology. In: 2020 IEEE International Conference on Big Data and Smart Computing (BigComp), pp. 526-530. IEEE","DOI":"10.1109\/BigComp48618.2020.00-12"},{"key":"16560_CR17","unstructured":"Devlin J, Gupta S, Girshick R, Mitchell M, Zitnick CL (2015) Exploring nearest neighbor approaches for image captioning. arXiv preprint arXiv:1505.04467"},{"key":"16560_CR18","doi-asserted-by":"crossref","unstructured":"Hessel J, Savva N, Wilber MJ (2015) Image representations and new domains in neural image captioning. arXiv preprint arXiv:1508.02091","DOI":"10.18653\/v1\/W15-2807"},{"key":"16560_CR19","unstructured":"Khan R, Islam, MS, Kanwal K, Iqbal M, Hossain M, Ye Z et al (2022) A deep neural framework for image caption generation using gru-based attention mechanism. arXiv preprint arXiv:2203.01594"},{"key":"16560_CR20","unstructured":"Kuznetsova P, Ordonez V, Berg A, Berg T, Choi Y (2012) Collective generation of natural image descriptions. In: Proceedings of the 50th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 359-368"},{"key":"16560_CR21","unstructured":"Mitchell M, Dodge J, Goyal A, Yamaguchi K, Stratos K, Han X, Mensch A, Berg A, Berg T, Daum\u00e9 III, H (2012) Midge: Generating image descriptions from computer vision detections. In: Proceedings of the 13th Conference of the European Chapter of the Association for Computational Linguistics, pp. 747-756"},{"key":"16560_CR22","doi-asserted-by":"publisher","first-page":"2693","DOI":"10.1609\/aaai.v34i03.5655","volume":"34","author":"PH Seo","year":"2020","unstructured":"Seo PH, Sharma P, Levinboim T, Han B, Soricut R (2020) Reinforcing an image caption generator using off-line human feedback. Proceedings of the AAAI Conference on Artificial Intelligence 34:2693\u20132700","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"16560_CR23","doi-asserted-by":"crossref","unstructured":"Zheng Y, Li Y, Wang S (2019) Intention oriented image captions with guiding objects. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8395-8404","DOI":"10.1109\/CVPR.2019.00859"},{"key":"16560_CR24","unstructured":"Mao J, Xu W, Yang Y,Wang J, Huang Z, Yuille A (2014) Deep captioning with multimodal recurrent neural networks (m-rnn). arXiv preprint arXiv:1412.6632"},{"key":"16560_CR25","doi-asserted-by":"crossref","unstructured":"Chen X, Lawrence Zitnick C (2015) Mind\u2019s eye: A recurrent visual representation for image caption generation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2422-2431","DOI":"10.1109\/CVPR.2015.7298856"},{"key":"16560_CR26","doi-asserted-by":"crossref","unstructured":"Karpathy A, Fei-Fei L (2015) Deep visual-semantic alignments for generating image descriptions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3128-3137","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"16560_CR27","doi-asserted-by":"crossref","unstructured":"Mathews A, Xie L, He X (2016) Senticap: Generating image descriptions with sentiments. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 30","DOI":"10.1609\/aaai.v30i1.10475"},{"key":"16560_CR28","doi-asserted-by":"crossref","unstructured":"Yao T, Pan Y, Li Y, Qiu Z, Mei T (2017) Boosting image captioning with attributes. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 4894-4902","DOI":"10.1109\/ICCV.2017.524"},{"key":"16560_CR29","unstructured":"Ilse M, Tomczak J, Welling M (2018) Attention-based deep multiple instance learning. In: International Conference on Machine Learning, pp. 2127-2136. PMLR"},{"key":"16560_CR30","doi-asserted-by":"crossref","unstructured":"Tanti M, Gatt A, Camilleri KP (2017) What is the role of recurrent neural networks (rnns) in an image caption generator? arXiv preprint arXiv:1708.02043","DOI":"10.18653\/v1\/W17-3506"},{"key":"16560_CR31","doi-asserted-by":"crossref","unstructured":"Aneja J, Deshpande A, Schwing AG (2018) Convolutional image captioning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5561-5570","DOI":"10.1109\/CVPR.2018.00583"},{"key":"16560_CR32","doi-asserted-by":"crossref","unstructured":"Guo L, Liu J, Tang J, Li J, Luo W, Lu H (2019) Aligning linguistic words and visual semantic units for image captioning. In: Proceedings of the 27th ACM International Conference on Multimedia, pp. 765-773","DOI":"10.1145\/3343031.3350943"},{"key":"16560_CR33","doi-asserted-by":"crossref","unstructured":"You Q, Jin H, Wang Z, Fang C, Luo J (2016) Image captioning with semantic attention. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4651-4659","DOI":"10.1109\/CVPR.2016.503"},{"key":"16560_CR34","doi-asserted-by":"publisher","first-page":"305","DOI":"10.1145\/3126686.3126717","volume":"2017","author":"L Zhou","year":"2017","unstructured":"Zhou L, Xu C, Koch P, Corso JJ (2017) Watch what you just said: Image captioning with text-conditional attention. Proceedings of the on Thematic Workshops of ACM Multimedia 2017:305\u2013313","journal-title":"Proceedings of the on Thematic Workshops of ACM Multimedia"},{"key":"16560_CR35","doi-asserted-by":"crossref","unstructured":"Jia X, Gavves E, Fernando B, Tuytelaars T (2015) Guiding the long-short term memory model for image caption generation. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2407-2415","DOI":"10.1109\/ICCV.2015.277"},{"key":"16560_CR36","doi-asserted-by":"crossref","unstructured":"Mun J, Cho M, Han B (2017) Text-guided attention model for image captioning. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 31","DOI":"10.1609\/aaai.v31i1.11237"},{"key":"16560_CR37","unstructured":"Xu K, Ba J, Kiros R, Cho K, Courville A, Salakhudinov R, Zemel R, Bengio Y (2015) Show, attend and tell: Neural image caption generation with visual attention. In: International Conference on Machine Learning, pp. 2048-2057. PMLR"},{"key":"16560_CR38","doi-asserted-by":"crossref","unstructured":"Anderson P, He X, Buehler C, Teney D, Johnson M, Gould S, Zhang L (2018) Bottom-up and top-down attention for image captioning and visual question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6077-6086","DOI":"10.1109\/CVPR.2018.00636"},{"key":"16560_CR39","doi-asserted-by":"crossref","unstructured":"Yao T, Pan Y, Li Y, Mei T (2018) Exploring visual relationship for image captioning. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 684-699","DOI":"10.1007\/978-3-030-01264-9_42"},{"key":"16560_CR40","doi-asserted-by":"publisher","first-page":"291","DOI":"10.1016\/j.neucom.2018.05.080","volume":"311","author":"S Bai","year":"2018","unstructured":"Bai S, An S (2018) A survey on automatic image caption generation. Neurocomputing 311:291\u2013304","journal-title":"Neurocomputing"},{"key":"16560_CR41","doi-asserted-by":"crossref","unstructured":"Janakiraman J, Unnikrishnan K (1992) A feedback model of visual attention. In: [Proceedings 1992] IJCNN International Joint Conference on Neural Networks, vol. 3, pp. 541-546. IEEE","DOI":"10.1109\/IJCNN.1992.227117"},{"issue":"2","key":"16560_CR42","doi-asserted-by":"publisher","first-page":"219","DOI":"10.1162\/089892904322984526","volume":"16","author":"MW Spratling","year":"2004","unstructured":"Spratling MW, Johnson MH (2004) A feedback model of visual attention. Journal of cognitive neuroscience 16(2):219\u2013237","journal-title":"Journal of cognitive neuroscience"},{"key":"16560_CR43","doi-asserted-by":"crossref","unstructured":"Pan Y, Yao T, Li Y, Mei T (2020) X-linear attention networks for image captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10971-10980","DOI":"10.1109\/CVPR42600.2020.01098"},{"key":"16560_CR44","doi-asserted-by":"crossref","unstructured":"Zhou Y, Wang M, Liu D, Hu Z, Zhang H (2020) More grounded image captioning by distilling image-text matching model. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4777-4786","DOI":"10.1109\/CVPR42600.2020.00483"},{"key":"16560_CR45","doi-asserted-by":"crossref","unstructured":"Lee K-H, Chen X, Hua G, Hu H, He X (2018) Stacked cross attention for imagetext matching. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 201-216","DOI":"10.1007\/978-3-030-01225-0_13"},{"key":"16560_CR46","doi-asserted-by":"crossref","unstructured":"Rennie SJ, Marcheret E, Mroueh Y, Ross J, Goel V (2017) Self-critical sequence training for image captioning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7008-7024","DOI":"10.1109\/CVPR.2017.131"},{"key":"16560_CR47","doi-asserted-by":"publisher","first-page":"2584","DOI":"10.1609\/aaai.v35i3.16361","volume":"35","author":"Z Song","year":"2021","unstructured":"Song Z, Zhou X, Mao Z, Tan J (2021) Image captioning with context-aware auxiliary guidance. Proceedings of the AAAI Conference on Artificial Intelligence 35:2584\u20132592","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"16560_CR48","doi-asserted-by":"crossref","unstructured":"Liu S, Zhu Z, Ye N, Guadarrama S, Murphy K (2017) Improved image captioning via policy gradient optimization of spider. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 873-881","DOI":"10.1109\/ICCV.2017.100"},{"key":"16560_CR49","unstructured":"Elliott D, Keller F (2013) Image description using visual dependency representations. In: Proceedings of the 2013 Conference on Empirical Methods in NaturalLanguage Processing, pp. 1292-1302"},{"key":"16560_CR50","doi-asserted-by":"publisher","first-page":"416","DOI":"10.1016\/j.neucom.2017.07.014","volume":"272","author":"P Kinghorn","year":"2018","unstructured":"Kinghorn P, Zhang L, Shao L (2018) A region-based image caption generator with refined descriptions. Neurocomputing 272:416\u2013424","journal-title":"Neurocomputing"},{"issue":"4","key":"16560_CR51","doi-asserted-by":"publisher","first-page":"419","DOI":"10.1016\/j.cviu.2009.03.008","volume":"114","author":"HJ Escalante","year":"2010","unstructured":"Escalante HJ, Hern\u00e1ndez CA, Gonzalez JA, L\u00f3pez-L\u00f3pez A, Montes M, Morales EF, Sucar LE, Villasenor L, Grubinger M (2010) The segmented and annotated iapr tc-12 benchmark. Computer vision and image understanding 114(4):419\u2013428","journal-title":"Computer vision and image understanding"},{"key":"16560_CR52","unstructured":"Lebret R, Pinheiro PO, Collobert R (2014) Simple image description generator via a linear phrase-based approach. arXiv preprint arXiv:1412.8419"},{"key":"16560_CR53","doi-asserted-by":"publisher","first-page":"86","DOI":"10.1016\/j.neucom.2018.12.026","volume":"333","author":"YH Tan","year":"2019","unstructured":"Tan YH, Chan CS (2019) Phrase-based image caption generator with hierarchical lstm network. Neurocomputing 333:86\u2013100","journal-title":"Neurocomputing"},{"key":"16560_CR54","doi-asserted-by":"crossref","unstructured":"Tan YH, Chan CS (2016) Phi-lstm: a phrase-based hierarchical lstm model for image captioning. In: Asian Conference on Computer Vision, pp. 101-117 Springer","DOI":"10.1007\/978-3-319-54193-8_7"},{"key":"16560_CR55","unstructured":"Van Miltenburg E (2016) Stereotyping and bias in the flickr30k dataset. arXiv preprint arXiv:1605.06083"},{"key":"16560_CR56","doi-asserted-by":"crossref","unstructured":"Lin T-Y, Maire M, Belongie S, Hays J, Perona P, Ramanan D, Doll\u00e1r P, Zitnick CL (2014) Microsoft coco: Common objects in context. In: European Conference on Computer Vision, pp. 740-755. Springer","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"16560_CR57","doi-asserted-by":"crossref","unstructured":"Anitha Kumari K, Mouneeshwari C, Udhaya R, Jasmitha R (2019) Automated image captioning for flickr8k dataset. In: International Conference on Artificial Intelligence, Smart Grid and Smart City Applications, pp. 679-687. Springer","DOI":"10.1007\/978-3-030-24051-6_62"},{"key":"16560_CR58","doi-asserted-by":"publisher","first-page":"207","DOI":"10.1162\/tacl_a_00177","volume":"2","author":"R Socher","year":"2014","unstructured":"Socher R, Karpathy A, Le QV, Manning CD, Ng AY (2014) Grounded compositional semantics for finding and describing images with sentences. Transactions of the Association for Computational Linguistics 2:207\u2013218","journal-title":"Transactions of the Association for Computational Linguistics"},{"key":"16560_CR59","doi-asserted-by":"publisher","first-page":"9627","DOI":"10.1109\/TIP.2020.3028651","volume":"29","author":"M Yang","year":"2020","unstructured":"Yang M, Liu J, Shen Y, Zhao Z, Chen X, Wu Q, Li C (2020) An ensemble of generation-and retrieval-based image captioning with dual generator generative adversarial network. IEEE Transactions on Image Processing 29:9627\u20139640","journal-title":"IEEE Transactions on Image Processing"},{"key":"16560_CR60","doi-asserted-by":"crossref","unstructured":"Feng Y, Ma L, Liu W, Luo J (2019) Unsupervised image captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4125-4134","DOI":"10.1109\/CVPR.2019.00425"},{"key":"16560_CR61","unstructured":"Kumar D, Gehani S, Oza P (2020) A review of deep learning based image captioning models"},{"key":"16560_CR62","doi-asserted-by":"crossref","unstructured":"Stefanini M, Cornia M, Baraldi L, Cascianelli S, Fiameni G, Cucchiara R (2022) From show to tell: A survey on deep learning-based image captioning. IEEE Transactions on Pattern Analysis and Machine Intelligence","DOI":"10.1109\/TPAMI.2022.3148210"},{"key":"16560_CR63","doi-asserted-by":"publisher","first-page":"13041","DOI":"10.1609\/aaai.v34i07.7005","volume":"34","author":"L Zhou","year":"2020","unstructured":"Zhou L, Palangi H, Zhang L, Hu H, Corso J, Gao J (2020) Unified visionlanguage pre-training for image captioning and vqa. Proceedings of the AAAI Conference on Artificial Intelligence 34:13041\u201313049","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"16560_CR64","doi-asserted-by":"crossref","unstructured":"Zhang P, Li X, Hu X, Yang J, Zhang L, Wang L, Choi Y, Gao J (2021) Vinvl: Revisiting visual representations in vision-language models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5579-5588","DOI":"10.1109\/CVPR46437.2021.00553"},{"key":"16560_CR65","doi-asserted-by":"crossref","unstructured":"Hu X, Gan Z, Wang J, Yang Z, Liu Z, Lu Y, Wang L (2022) Scaling up visionlanguage pre-training for image captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 17980-17989","DOI":"10.1109\/CVPR52688.2022.01745"},{"key":"16560_CR66","doi-asserted-by":"crossref","unstructured":"He S, Liao W, Tavakoli HR, Yang M, Rosenhahn B, Pugeault N (2020) Image captioning through image transformer. In: Proceedings of the Asian Conference on Computer Vision","DOI":"10.1007\/978-3-030-69538-5_10"},{"key":"16560_CR67","doi-asserted-by":"crossref","unstructured":"Cornia M, Stefanini M, Baraldi L, Cucchiara R (2020) Meshed-memory transformer for image captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10578-10587","DOI":"10.1109\/CVPR42600.2020.01059"},{"key":"16560_CR68","doi-asserted-by":"crossref","unstructured":"Li G, Zhu L, Liu P, Yang Y (2019) Entangled transformer for image captioning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8928-8937","DOI":"10.1109\/ICCV.2019.00902"},{"issue":"12","key":"16560_CR69","doi-asserted-by":"publisher","first-page":"4467","DOI":"10.1109\/TCSVT.2019.2947482","volume":"30","author":"J Yu","year":"2019","unstructured":"Yu J, Li J, Yu Z, Huang Q (2019) Multimodal transformer with multi-view visual representation for image captioning. IEEE Trans Circ Syst Video Technol 30(12):4467\u20134480","journal-title":"IEEE Trans Circ Syst Video Technol"},{"key":"16560_CR70","doi-asserted-by":"crossref","unstructured":"Xiong Y, Du B, Yan P (2019) Reinforced transformer for medical image captioning. In: Machine Learning in Medical Imaging: 10th International Workshop, MLMI 2019, Held in Conjunction with MICCAI 2019, Shenzhen, China, October 13, 2019, Proceedings 10, pp. 673-680. Springer","DOI":"10.1007\/978-3-030-32692-0_77"},{"key":"16560_CR71","doi-asserted-by":"publisher","first-page":"285","DOI":"10.1016\/j.patcog.2019.01.028","volume":"90","author":"X Xiao","year":"2019","unstructured":"Xiao X, Wang L, Ding K, Xiang S, Pan C (2019) Dense semantic embedding network for image captioning. Pattern Recognition 90:285\u2013296","journal-title":"Pattern Recognition"},{"key":"16560_CR72","doi-asserted-by":"crossref","unstructured":"Kim D-J, Choi J, Oh T-H, Kweon IS (2019) Dense relational captioning: Triple-stream networks for relationship-based captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6271-6280","DOI":"10.1109\/CVPR.2019.00643"},{"issue":"11","key":"16560_CR73","doi-asserted-by":"publisher","first-page":"7348","DOI":"10.1109\/TPAMI.2021.3119754","volume":"44","author":"D-J Kim","year":"2021","unstructured":"Kim D-J, Oh T-H, Choi J, Kweon IS (2021) Dense relational image captioning via multi-task triple-stream networks. IEEE Transactions on Pattern Analysis and Machine Intelligence 44(11):7348\u20137362","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"16560_CR74","doi-asserted-by":"crossref","unstructured":"Johnson J, Karpathy A, Fei-Fei L (2016) Densecap: Fully convolutional localization networks for dense captioning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4565-4574","DOI":"10.1109\/CVPR.2016.494"},{"key":"16560_CR75","doi-asserted-by":"crossref","unstructured":"Li L, Gan Z, Cheng Y, Liu J (2019) Relation-aware graph attention network for visual question answering. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10313-10322","DOI":"10.1109\/ICCV.2019.01041"},{"key":"16560_CR76","doi-asserted-by":"crossref","unstructured":"Shao Z, Han J, Debattista K, Pang Y (2023) Textual context-aware dense captioning with diverse words. IEEE Transactions on Multimedia","DOI":"10.1109\/TMM.2023.3241517"},{"key":"16560_CR77","unstructured":"Shao Z, Han J, Marnerides D, Debattista K (2022) Region-object relation-aware dense captioning via transformer. IEEE Transactions on Neural Networks and Learning Systems"},{"key":"16560_CR78","doi-asserted-by":"crossref","unstructured":"Sharma G, Kalena P, Malde N, Nair A, Parkar S (2019) Visual image caption generator using deep learning. In: 2nd International Conference on Advances in Science & Technology (ICAST)","DOI":"10.2139\/ssrn.3368837"},{"key":"16560_CR79","doi-asserted-by":"crossref","unstructured":"Hendricks LA, Venugopalan S, Rohrbach M, Mooney R, Saenko K, Darrell T (2016) Deep compositional captioning: Describing novel object categories without paired training data. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1-10","DOI":"10.1109\/CVPR.2016.8"},{"key":"16560_CR80","doi-asserted-by":"crossref","unstructured":"Venugopalan S, Anne Hendricks L, Rohrbach M, Mooney R, Darrell T, Saenko K (2017) Captioning images with diverse objects. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5753-5761","DOI":"10.1109\/CVPR.2017.130"},{"key":"16560_CR81","doi-asserted-by":"crossref","unstructured":"Chen J, Guo H, Yi K, Li B, Elhoseiny M (2022) Visualgpt: Data-efficient adaptation of pretrained language models for image captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18030-18040","DOI":"10.1109\/CVPR52688.2022.01750"},{"key":"16560_CR82","doi-asserted-by":"crossref","unstructured":"Sharma P, Ding N, Goodman S, Soricut R (2018) Conceptual captions: A cleaned, hypernymed, image alt-text dataset for automatic image captioning. In: Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 2556-2565","DOI":"10.18653\/v1\/P18-1238"},{"issue":"2","key":"16560_CR83","doi-asserted-by":"publisher","first-page":"304","DOI":"10.1093\/jamia\/ocv080","volume":"23","author":"D Demner-Fushman","year":"2016","unstructured":"Demner-Fushman D, Kohli MD, Rosenman MB, Shooshan SE, Rodriguez L, Antani S, Thoma GR, McDonald CJ (2016) Preparing a collection of radiology examinations for distribution and retrieval. J Am Med Inf Assoc 23(2):304\u2013310","journal-title":"J Am Med Inf Assoc"},{"key":"16560_CR84","doi-asserted-by":"crossref","unstructured":"Li X, Yin X, Li C, Zhang P, Hu X, Zhang L, Wang L, Hu H, Dong L, Wei F et al (2020) Oscar: Object-semantics aligned pre-training for visionlanguage tasks. In: European Conference on Computer Vision, pp. 121-137. Springer","DOI":"10.1007\/978-3-030-58577-8_8"},{"key":"16560_CR85","doi-asserted-by":"crossref","unstructured":"Hu X, Yin X, Lin K, Wang L, Zhang L, Gao J, Liu Z (2020) Vivo: Visual vocabulary pre-training for novel object captioning. arXiv preprint arXiv:2009.13682","DOI":"10.1609\/aaai.v35i2.16249"},{"key":"16560_CR86","doi-asserted-by":"crossref","unstructured":"Gonog L, Zhou Y (2019) A review: generative adversarial networks. In: 2019 14th IEEE Conference on Industrial Electronics and Applications (ICIEA), pp. 505- 510. IEEE","DOI":"10.1109\/ICIEA.2019.8833686"},{"key":"16560_CR87","doi-asserted-by":"publisher","first-page":"8142","DOI":"10.1609\/aaai.v33i01.33018142","volume":"33","author":"C Chen","year":"2019","unstructured":"Chen C, Mu S, Xiao W, Ye Z, Wu L, Ju Q (2019) Improving image captioning with conditional generative adversarial nets. Proceedings of the AAAI Conference on Artificial Intelligence 33:8142\u20138150","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"16560_CR88","doi-asserted-by":"publisher","first-page":"8626","DOI":"10.1609\/aaai.v33i01.33018626","volume":"33","author":"N Li","year":"2019","unstructured":"Li N, Chen Z, Liu S (2019) Meta learning for image captioning. Proceedings of the AAAI Conference on Artificial Intelligence 33:8626\u20138633","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"16560_CR89","doi-asserted-by":"publisher","first-page":"2286","DOI":"10.1609\/aaai.v35i3.16328","volume":"35","author":"Y Luo","year":"2021","unstructured":"Luo Y, Ji J, Sun X, Cao L, Wu Y, Huang F, Lin C-W, Ji R (2021) Dual-level collaborative transformer for image captioning. Proceedings of the AAAI Conference on Artificial Intelligence 35:2286\u20132293","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"16560_CR90","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1162\/tacl_a_00166","volume":"2","author":"P Young","year":"2014","unstructured":"Young P, Lai A, Hodosh M, Hockenmaier J (2014) From image descriptions to visual denotations: New similarity metrics for semantic inference over event descriptions. Transactions of the Association for Computational Linguistics 2:67\u201378","journal-title":"Transactions of the Association for Computational Linguistics"},{"key":"16560_CR91","doi-asserted-by":"crossref","unstructured":"Agrawal H, Desai K, Wang Y, Chen X, Jain R, Johnson M, Batra D, Parikh D, Lee S, Anderson P (2019) Nocaps: Novel object captioning at scale. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8948-8957","DOI":"10.1109\/ICCV.2019.00904"},{"key":"16560_CR92","doi-asserted-by":"crossref","unstructured":"Yoshikawa Y, Shigeto Y, Takeuchi A (2017) Stair captions: Constructing a largescale japanese image caption dataset. arXiv preprint arXiv:1705.00823","DOI":"10.18653\/v1\/P17-2066"},{"key":"16560_CR93","doi-asserted-by":"crossref","unstructured":"Hsu T-Y, Giles CL, Huang T-H (2021) Scicap: Generating captions for scientific figures. arXiv preprint arXiv:2110.11624","DOI":"10.18653\/v1\/2021.findings-emnlp.277"},{"key":"16560_CR94","doi-asserted-by":"crossref","unstructured":"Sidorov O, Hu R, Rohrbach M, Singh A (2020) Textcaps: a dataset for image captioning with reading comprehension. In: European Conference on Computer Vision, pp. 742-758. Springer","DOI":"10.1007\/978-3-030-58536-5_44"},{"key":"16560_CR95","doi-asserted-by":"crossref","unstructured":"Mao J, Huang J, Toshev A, Camburu O, Yuille AL, Murphy K (2016) Generation and comprehension of unambiguous object descriptions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 11-20","DOI":"10.1109\/CVPR.2016.9"},{"key":"16560_CR96","doi-asserted-by":"crossref","unstructured":"Changpinyo S, Sharma P, Ding N, Soricut R (2021) Conceptual 12m: Pushing web-scale image-text pre-training to recognize long-tail visual concepts. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3558-3568","DOI":"10.1109\/CVPR46437.2021.00356"},{"key":"16560_CR97","doi-asserted-by":"crossref","unstructured":"Papineni K, Roukos S, Ward T, Zhu W-J (2002) Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics, pp. 311-318","DOI":"10.3115\/1073083.1073135"},{"key":"16560_CR98","doi-asserted-by":"crossref","unstructured":"Denkowski M, Lavie A (2014) Meteor universal: Language specific translation evaluation for any target language. In: Proceedings of the Ninth Workshop on Statistical Machine Translation, pp. 376-380","DOI":"10.3115\/v1\/W14-3348"},{"key":"16560_CR99","unstructured":"Lin C-Y (2004) Rouge: A package for automatic evaluation of summaries. In: Text Summarization Branches Out, pp. 74-81"},{"key":"16560_CR100","doi-asserted-by":"crossref","unstructured":"Vedantam R, Lawrence Zitnick C, Parikh D (2015) Cider: Consensus-based image description evaluation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4566-4575","DOI":"10.1109\/CVPR.2015.7299087"},{"issue":"2","key":"16560_CR101","doi-asserted-by":"publisher","first-page":"2088","DOI":"10.1109\/TPAMI.2022.3159811","volume":"45","author":"J Wang","year":"2022","unstructured":"Wang J, Xu W, Wang Q, Chan AB (2022) On distinctive image captioning via comparing and reweighting. IEEE Transactions on Pattern Analysis and Machine Intelligence 45(2):2088\u20132103","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"16560_CR102","doi-asserted-by":"crossref","unstructured":"Anderson P, Fernando B, Johnson M, Gould S (2016) Spice: Semantic propositional image caption evaluation. In: European Conference on Computer Vision, pp. 382-398. Springer","DOI":"10.1007\/978-3-319-46454-1_24"},{"key":"16560_CR103","unstructured":"Sundaramoorthy C, Kelvin LZ, Sarin M, Gupta S (2021) End-to-end attentionbased image captioning. arXiv preprint arXiv:2104.14721"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-16560-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-023-16560-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-16560-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,27]],"date-time":"2024-10-27T07:16:04Z","timestamp":1730013364000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-023-16560-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,8,31]]},"references-count":103,"journal-issue":{"issue":"9","published-online":{"date-parts":[[2024,3]]}},"alternative-id":["16560"],"URL":"https:\/\/doi.org\/10.1007\/s11042-023-16560-x","relation":{},"ISSN":["1573-7721"],"issn-type":[{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,8,31]]},"assertion":[{"value":"9 May 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 July 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 August 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"31 August 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of Interest"}}]}}