{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,17]],"date-time":"2026-04-17T23:49:09Z","timestamp":1776469749649,"version":"3.51.2"},"reference-count":96,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2024,3,1]],"date-time":"2024-03-01T00:00:00Z","timestamp":1709251200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,3,1]],"date-time":"2024-03-01T00:00:00Z","timestamp":1709251200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"NSFC","doi-asserted-by":"crossref","award":["61771386"],"award-info":[{"award-number":["61771386"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100015401","name":"Key Research and Development Program of Shaanxi","doi-asserted-by":"crossref","award":["2020SF-359"],"award-info":[{"award-number":["2020SF-359"]}],"id":[{"id":"10.13039\/501100015401","id-type":"DOI","asserted-by":"crossref"}]},{"name":"Key Lab. of Manufacturing Equipment of Shaanxi Province","award":["JXZZZB-2022-02"],"award-info":[{"award-number":["JXZZZB-2022-02"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2024,3]]},"DOI":"10.1007\/s10489-024-05389-y","type":"journal-article","created":{"date-parts":[[2024,3,25]],"date-time":"2024-03-25T10:02:21Z","timestamp":1711360941000},"page":"4300-4318","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["Weakly supervised grounded image captioning with semantic matching"],"prefix":"10.1007","volume":"54","author":[{"given":"Sen","family":"Du","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hong","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guangfeng","family":"Lin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuanyuan","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dong","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jing","family":"Shi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhong","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,3,25]]},"reference":[{"issue":"1","key":"5389_CR1","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1007\/s11263-016-0981-7","volume":"123","author":"R Krishna","year":"2017","unstructured":"Krishna R, Zhu Y, Groth O, Johnson J, Hata K, Kravitz J, Chen S, Kalantidis Y, Li L-J, Shamma DA et al (2017) Visual genome: Connecting language and vision using crowdsourced dense image annotations. Int J Comput Vis 123(1):32\u201373","journal-title":"Int J Comput Vis"},{"key":"5389_CR2","doi-asserted-by":"publisher","first-page":"33","DOI":"10.1016\/j.cviu.2017.12.004","volume":"173","author":"S Aditya","year":"2018","unstructured":"Aditya S, Yang Y, Baral C, Aloimonos Y, Ferm\u00fcller C (2018) Image understanding using vision and reasoning through scene description graph. Comput Vis Image Understanding 173:33\u201345","journal-title":"Comput Vis Image Understanding"},{"key":"5389_CR3","doi-asserted-by":"crossref","unstructured":"Farhadi A, Hejrati M, Sadeghi MA, Young P, Rashtchian C, Hockenmaier J, Forsyth D (2010) Every picture tells a story: Generating sentences from images. In: European conference on computer vision, pp 15\u201329. Springer","DOI":"10.1007\/978-3-642-15561-1_2"},{"issue":"12","key":"5389_CR4","doi-asserted-by":"publisher","first-page":"2891","DOI":"10.1109\/TPAMI.2012.162","volume":"35","author":"G Kulkarni","year":"2013","unstructured":"Kulkarni G, Premraj V, Ordonez V, Dhar S, Li S, Choi Y, Berg AC, Berg TL (2013) Babytalk: Understanding and generating simple image descriptions. IEEE Trans Pattern Anal Mach Intell 35(12):2891\u20132903","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"5389_CR5","doi-asserted-by":"crossref","unstructured":"Hendricks LA, Burns K, Saenko K, Darrell T, Rohrbach A (2018) Women also snowboard: Overcoming bias in captioning models. In: Proceedings of the european conference on computer vision (ECCV), pp 771\u2013787","DOI":"10.1007\/978-3-030-01219-9_47"},{"key":"5389_CR6","first-page":"1865","volume":"33","author":"F Liu","year":"2020","unstructured":"Liu F, Ren X, Wu X, Ge S, Fan W, Zou Y, Sun X (2020) Prophet attention: Predicting attention with future attention. Adv Neural Inf Process Syst 33:1865\u20131876","journal-title":"Adv Neural Inf Process Syst"},{"key":"5389_CR7","doi-asserted-by":"crossref","unstructured":"Devlin J, Cheng H, Fang H, Gupta S, Deng L, He X, Zweig G, Mitchell M (2015) Language models for image captioning: The quirks and what works. arXiv:1505.01809","DOI":"10.3115\/v1\/P15-2017"},{"key":"5389_CR8","doi-asserted-by":"crossref","unstructured":"Liu X, Li H, Shao J,Chen D, Wang X Show, tell and discriminate: Image captioning by self-retrieval with partially labeled data. In: Proceedings of the european conference on computer vision (ECCV), pp 338\u2013354 (2018)","DOI":"10.1007\/978-3-030-01267-0_21"},{"key":"5389_CR9","unstructured":"Ordonez V, Kulkarni G, Berg T (2011) Im2text: Describing images using 1 million captioned photographs. Adv Neural Inf Process Syst 24"},{"key":"5389_CR10","doi-asserted-by":"crossref","unstructured":"Lu J, Yang J, Batra D, Parikh D (2018) Neural baby talk. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 7219\u20137228","DOI":"10.1109\/CVPR.2018.00754"},{"key":"5389_CR11","unstructured":"Mitchell M, Dodge J, Goyal A, Yamaguchi K, Stratos K, Han X, Mensch A, Berg A, Berg T, Daum\u00e9-III H (2012) Midge: Generating image descriptions from computer vision detections. In: Proceedings of the 13th conference of the european chapter of the association for computational linguistics, pp 747\u2013756"},{"key":"5389_CR12","unstructured":"Xu K, Ba J, Kiros R, Cho K, Courville A, Salakhudinov R, Zemel R, Bengio Y (2015) Show, attend and tell: Neural image caption generation with visual attention. In: International conference on machine learning, pp 2048\u20132057. PMLR"},{"key":"5389_CR13","doi-asserted-by":"crossref","unstructured":"Chen L, Zhang H, Xiao J, Nie L, Shao J, Liu W, Chua T-S (2017) Sca-cnn: Spatial and channel-wise attention in convolutional networks for image captioning. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 5659\u20135667","DOI":"10.1109\/CVPR.2017.667"},{"key":"5389_CR14","doi-asserted-by":"crossref","unstructured":"Anderson P, He X, Buehler C, Teney D, Johnson M, Gould S, Zhang L Bottom-up and top-down attention for image captioning and visual question answering. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 6077\u20136086 (2018)","DOI":"10.1109\/CVPR.2018.00636"},{"key":"5389_CR15","doi-asserted-by":"publisher","first-page":"7615","DOI":"10.1109\/TIP.2020.3004729","volume":"29","author":"J Ji","year":"2020","unstructured":"Ji J, Xu C, Zhang X, Wang B, Song X (2020) Spatio-temporal memory attention for image captioning. IEEE Trans Image Process 29:7615\u20137628","journal-title":"IEEE Trans Image Process"},{"key":"5389_CR16","first-page":"8957","volume":"33","author":"W Wang","year":"2019","unstructured":"Wang W, Chen Z, Hu H (2019) Hierarchical attention network for image captioning. Proc AAAI Conf Artif Intell 33:8957\u20138964","journal-title":"Proc AAAI Conf Artif Intell"},{"key":"5389_CR17","doi-asserted-by":"publisher","first-page":"1775","DOI":"10.1109\/TMM.2021.3072479","volume":"24","author":"L Yu","year":"2021","unstructured":"Yu L, Zhang J, Wu Q (2021) Dual attention on pyramid feature maps for image captioning. IEEE Trans Multimed 24:1775\u20131786","journal-title":"IEEE Trans Multimed"},{"key":"5389_CR18","first-page":"2584","volume":"35","author":"Z Song","year":"2021","unstructured":"Song Z, Zhou X, Mao Z, Tan J (2021) Image captioning with context-aware auxiliary guidance. Proc AAAI Conf Artif Intell 35:2584\u20132592","journal-title":"Proc AAAI Conf Artif Intell"},{"key":"5389_CR19","doi-asserted-by":"crossref","unstructured":"Pan Y, Yao T, Li Y, Mei T (2020) X-linear attention networks for image captioning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 10971\u201310980","DOI":"10.1109\/CVPR42600.2020.01098"},{"key":"5389_CR20","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser \u0141, Polosukhin I (2017) Attention is all you need. Adv Neural Inf Process Syst 30"},{"key":"5389_CR21","doi-asserted-by":"crossref","unstructured":"Huang L, Wang W, Chen J, Wei X-Y (2019) Attention on attention for image captioning. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 4634\u20134643","DOI":"10.1109\/ICCV.2019.00473"},{"key":"5389_CR22","doi-asserted-by":"crossref","unstructured":"Cornia M, Stefanini M, Baraldi L, Cucchiara R (2020) Meshed-memory transformer for image captioning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 10578\u201310587","DOI":"10.1109\/CVPR42600.2020.01059"},{"issue":"11","key":"5389_CR23","doi-asserted-by":"publisher","first-page":"7706","DOI":"10.1109\/TCSVT.2022.3181490","volume":"32","author":"W Jiang","year":"2022","unstructured":"Jiang W, Zhou W, Hu H (2022) Double-stream position learning transformer network for image captioning. IEEE Trans Circ Syst Vid Technol 32(11):7706\u20137718. https:\/\/doi.org\/10.1109\/TCSVT.2022.3181490","journal-title":"IEEE Trans Circ Syst Vid Technol"},{"key":"5389_CR24","first-page":"2286","volume":"35","author":"Y Luo","year":"2021","unstructured":"Luo Y, Ji J, Sun X, Cao L, Wu Y, Huang F, Lin C-W, Ji R (2021) Dual-level collaborative transformer for image captioning. Proc AAAI Conf Artif Intell 35:2286\u20132293","journal-title":"Proc AAAI Conf Artif Intell"},{"key":"5389_CR25","first-page":"13041","volume":"34","author":"L Zhou","year":"2020","unstructured":"Zhou L, Palangi H, Zhang L, Hu H, Corso J, Gao J (2020) Unified vision-language pre-training for image captioning and vqa. Proc AAAI Conf Artif Intell 34:13041\u201313049","journal-title":"Proc AAAI Conf Artif Intell"},{"key":"5389_CR26","first-page":"8518","volume":"35","author":"Y Li","year":"2021","unstructured":"Li Y, Pan Y, Yao T, Chen J, Mei T (2021) Scheduled sampling in vision-language pretraining with decoupled encoder-decoder network. Proc AAAI Conf Artif Intell 35:8518\u20138526","journal-title":"Proc AAAI Conf Artif Intell"},{"key":"5389_CR27","doi-asserted-by":"crossref","unstructured":"Zhang P, Li X, Hu X, Yang J, Zhang L, Wang L, Choi Y, Gao J (2021) Vinvl: Revisiting visual representations in vision-language models. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 5579\u20135588","DOI":"10.1109\/CVPR46437.2021.00553"},{"key":"5389_CR28","doi-asserted-by":"crossref","unstructured":"Lu J, Goswami V, Rohrbach M, Parikh D, Lee S (2020) 12-in-1: Multi-task vision and language representation learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 10437\u201310446","DOI":"10.1109\/CVPR42600.2020.01045"},{"key":"5389_CR29","doi-asserted-by":"crossref","unstructured":"Li X, Yin X, Li C, Zhang P, Hu X, Zhang L, Wang L, Hu H, Dong L, Wei F et al (2020) Oscar: Object-semantics aligned pre-training for vision-language tasks. In: European conference on computer vision, pp 121\u2013137. Springer","DOI":"10.1007\/978-3-030-58577-8_8"},{"key":"5389_CR30","unstructured":"Wang Z, Yu J, Yu AW, Dai Z, Tsvetkov Y, Cao Y (2021) Simvlm: Simple visual language model pretraining with weak supervision. arXiv:2108.10904"},{"key":"5389_CR31","doi-asserted-by":"crossref","unstructured":"You Q, Jin H, Wang Z, Fang C, Luo J (2016) Image captioning with semantic attention. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 4651\u20134659","DOI":"10.1109\/CVPR.2016.503"},{"key":"5389_CR32","doi-asserted-by":"crossref","unstructured":"Gan Z, Gan C, He X, Pu Y, Tran K, Gao J, Carin L, Deng L (2017) Semantic compositional networks for visual captioning. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 5630\u20135639","DOI":"10.1109\/CVPR.2017.127"},{"key":"5389_CR33","doi-asserted-by":"crossref","unstructured":"Dai B, Fidler S, Urtasun R, Lin D (2017) Towards diverse and natural image descriptions via a conditional gan. In: Proceedings of the IEEE international conference on computer vision, pp 2970\u20132979","DOI":"10.1109\/ICCV.2017.323"},{"key":"5389_CR34","doi-asserted-by":"crossref","unstructured":"Shetty R, Rohrbach M, Anne-Hendricks L, Fritz M, Schiele B (2017) Speaking the same language: Matching machine to human captions by adversarial training. In: Proceedings of the IEEE international conference on computer vision, pp 4135\u20134144","DOI":"10.1109\/ICCV.2017.445"},{"key":"5389_CR35","doi-asserted-by":"crossref","unstructured":"Dognin P, Melnyk I, Mroueh Y, Ross J, Sercu T (2019) Adversarial semantic alignment for improved image captions. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 10463\u201310471","DOI":"10.1109\/CVPR.2019.01071"},{"key":"5389_CR36","doi-asserted-by":"publisher","first-page":"92","DOI":"10.1109\/TMM.2020.2976552","volume":"23","author":"J Zhang","year":"2020","unstructured":"Zhang J, Mei K, Zheng Y, Fan J (2020) Integrating part of speech guidance for image captioning. IEEE Trans Multimed 23:92\u2013104","journal-title":"IEEE Trans Multimed"},{"issue":"1","key":"5389_CR37","doi-asserted-by":"publisher","first-page":"43","DOI":"10.1109\/TCSVT.2021.3067449","volume":"32","author":"C Yan","year":"2021","unstructured":"Yan C, Hao Y, Li L, Yin J, Liu A, Mao Z, Chen Z, Gao X (2021) Task-adaptive attention for image captioning. IEEE Trans Circ Syst Vid Technol 32(1):43\u201351","journal-title":"IEEE Trans Circ Syst Vid Technol"},{"key":"5389_CR38","doi-asserted-by":"crossref","unstructured":"Yang X, Zhang H, Gao C, Cai J (2022) Learning to collocate visual-linguistic neural modules for image captioning. Int J Comput Vis 1\u201319","DOI":"10.1007\/s11263-022-01692-8"},{"key":"5389_CR39","doi-asserted-by":"crossref","unstructured":"Guo L, Liu J, Tang J, Li J, Luo W, Lu H (2019) Aligning linguistic words and visual semantic units for image captioning. In: Proceedings of the 27th ACM international conference on multimedia, pp 765\u2013773","DOI":"10.1145\/3343031.3350943"},{"key":"5389_CR40","doi-asserted-by":"publisher","first-page":"107928","DOI":"10.1016\/j.patcog.2021.107928","volume":"115","author":"J Ji","year":"2021","unstructured":"Ji J, Du Z, Zhang X (2021) Divergent-convergent attention for image captioning. Pattern Recognit 115:107928","journal-title":"Pattern Recognit"},{"key":"5389_CR41","unstructured":"Milewski V, Moens M-F, Calixto I (2020) Are scene graphs good enough to improve image captioning? arXiv:2009.12313"},{"key":"5389_CR42","first-page":"3394","volume":"35","author":"W Zhang","year":"2021","unstructured":"Zhang W, Shi H, Tang S, Xiao J, Yu Q, Zhuang Y (2021) Consensus graph representation learning for better grounded image captioning. Proc AAAI Conf Artif Intell 35:3394\u20133402","journal-title":"Proc AAAI Conf Artif Intell"},{"key":"5389_CR43","doi-asserted-by":"crossref","unstructured":"Chen N, Pan X, Chen R, Yang L, Lin Z, Ren Y, Yuan H, Guo X, Huang F, Wang W (2021) Distributed attention for grounded image captioning. In: Proceedings of the 29th ACM international conference on multimedia, pp 1966\u20131975","DOI":"10.1145\/3474085.3475354"},{"key":"5389_CR44","doi-asserted-by":"crossref","unstructured":"Zhou Y, Wang M, Liu D, Hu Z, Zhang H (2020) More grounded image captioning by distilling image-text matching model. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 4777\u20134786","DOI":"10.1109\/CVPR42600.2020.00483"},{"key":"5389_CR45","doi-asserted-by":"crossref","unstructured":"Hu W, Wang L, Xu L (2022) Spatial-semantic attention for grounded image captioning. In: 2022 IEEE international conference on image processing (ICIP), pp 61\u201365. IEEE","DOI":"10.1109\/ICIP46576.2022.9897578"},{"key":"5389_CR46","doi-asserted-by":"crossref","unstructured":"Jiang W, Zhu M, Fang Y, Shi G, Zhao X, Liu Y (2022) Visual cluster grounding for image captioning. IEEE Trans Image Process","DOI":"10.1109\/TIP.2022.3177318"},{"key":"5389_CR47","doi-asserted-by":"crossref","unstructured":"Zhang H, Niu Y, Chang S-F (2018) Grounding referring expressions in images by variational context. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 4158\u20134166","DOI":"10.1109\/CVPR.2018.00437"},{"issue":"2","key":"5389_CR48","doi-asserted-by":"publisher","first-page":"394","DOI":"10.1109\/TPAMI.2018.2797921","volume":"41","author":"L Wang","year":"2018","unstructured":"Wang L, Li Y, Huang J, Lazebnik S (2018) Learning two-branch neural networks for image-text matching tasks. IEEE Trans Pattern Anal Mach Intell 41(2):394\u2013407","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"5389_CR49","first-page":"11645","volume":"34","author":"Y Liu","year":"2020","unstructured":"Liu Y, Wan B, Zhu X, He X (2020) Learning cross-modal context graph for visual grounding. Proc AAAI Conf Artif Intell 34:11645\u201311652","journal-title":"Proc AAAI Conf Artif Intell"},{"key":"5389_CR50","doi-asserted-by":"crossref","unstructured":"Yang Z, Gong B, Wang L, Huang W, Yu D, Luo J (2019) A fast and accurate one-stage approach to visual grounding. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 4683\u20134693","DOI":"10.1109\/ICCV.2019.00478"},{"key":"5389_CR51","doi-asserted-by":"crossref","unstructured":"Huang B, Lian D, Luo W, Gao S (2021) Look before you leap: Learning landmark features for one-stage visual grounding. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 16888\u201316897","DOI":"10.1109\/CVPR46437.2021.01661"},{"key":"5389_CR52","doi-asserted-by":"crossref","unstructured":"Rohrbach A, Rohrbach M, Hu R, Darrell T, Schiele B Grounding of textual phrases in images by reconstruction. In: European conference on computer vision, pp 817\u2013834 (2016). Springer","DOI":"10.1007\/978-3-319-46448-0_49"},{"key":"5389_CR53","doi-asserted-by":"crossref","unstructured":"Chen K, Gao J, Nevatia R (2018) Knowledge aided consistency for weakly supervised phrase grounding. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 4042\u20134050","DOI":"10.1109\/CVPR.2018.00425"},{"key":"5389_CR54","doi-asserted-by":"crossref","unstructured":"Wang L, Huang J, Li Y, Xu K, Yang Z, Yu D (2021) Improving weakly supervised visual grounding by contrastive knowledge distillation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 14090\u201314100","DOI":"10.1109\/CVPR46437.2021.01387"},{"key":"5389_CR55","doi-asserted-by":"crossref","unstructured":"Gupta T, Vahdat A, Chechik G, Yang X, Kautz J, Hoiem D (2020) Contrastive learning for weakly supervised phrase grounding. In: European conference on computer vision, pp 752\u2013768. Springer","DOI":"10.1007\/978-3-030-58580-8_44"},{"key":"5389_CR56","doi-asserted-by":"crossref","unstructured":"Liu A-A, Zhai Y, Xu N, Nie W, Li W, Zhang Y (2021) Region-aware image captioning via interaction learning. IEEE Trans Circ Syst Vid Technol","DOI":"10.1109\/TCSVT.2021.3107035"},{"key":"5389_CR57","doi-asserted-by":"crossref","unstructured":"Zhou L, Kalantidis Y, Chen X, Corso JJ, Rohrbach M (2019) Grounded video description. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 6578\u20136587","DOI":"10.1109\/CVPR.2019.00674"},{"issue":"1","key":"5389_CR58","doi-asserted-by":"publisher","first-page":"52","DOI":"10.1109\/TCSVT.2021.3063297","volume":"32","author":"Y Bin","year":"2022","unstructured":"Bin Y, Ding Y, Peng B, Peng L, Yang Y, Chua T-S (2022) Entity slot filling for visual captioning. IEEE Trans Circ Syst Vid Technol 32(1):52\u201362. https:\/\/doi.org\/10.1109\/TCSVT.2021.3063297","journal-title":"IEEE Trans Circ Syst Vid Technol"},{"key":"5389_CR59","doi-asserted-by":"crossref","unstructured":"Ma C-Y, Kalantidis Y, AlRegib G, Vajda P, Rohrbach M, Kira Z Learning to generate grounded visual captions without localization supervision. In: European conference on computer vision, pp 353\u2013370 (2020). Springer","DOI":"10.1007\/978-3-030-58523-5_21"},{"key":"5389_CR60","unstructured":"Radford A, Kim JW, Hallacy C, Ramesh A, Goh G, Agarwal S, Sastry G, Askell A, Mishkin P, Clark J, et al (2021) Learning transferable visual models from natural language supervision. In: International conference on machine learning, pp 8748\u20138763. PMLR"},{"issue":"7","key":"5389_CR61","doi-asserted-by":"publisher","first-page":"1956","DOI":"10.1007\/s11263-020-01316-z","volume":"128","author":"A Kuznetsova","year":"2020","unstructured":"Kuznetsova A, Rom H, Alldrin N, Uijlings J, Krasin I, Pont-Tuset J, Kamali S, Popov S, Malloci M, Kolesnikov A et al (2020) The open images dataset v4: Unified image classification, object detection, and visual relationship detection at scale. Int J Comput Vis 128(7):1956\u20131981","journal-title":"Int J Comput Vis"},{"key":"5389_CR62","doi-asserted-by":"crossref","unstructured":"Rennie SJ, Marcheret E, Mroueh Y, Ross J, Goel V (2017) Self-critical sequence training for image captioning. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 7008\u20137024","DOI":"10.1109\/CVPR.2017.131"},{"key":"5389_CR63","doi-asserted-by":"crossref","unstructured":"Luo R, Price B, Cohen S, Shakhnarovich G (2018) Discriminability objective for training descriptive captions.In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 6964\u20136974","DOI":"10.1109\/CVPR.2018.00728"},{"key":"5389_CR64","doi-asserted-by":"crossref","unstructured":"Lin T-Y, Maire M, Belongie S, Hays J, Perona P, Ramanan D, Doll\u00e1r P, Zitnick CL (2014) Microsoft coco: Common objects in context. In: European conference on computer vision, pp 740\u2013755. Springer","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"5389_CR65","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1162\/tacl_a_00166","volume":"2","author":"P Young","year":"2014","unstructured":"Young P, Lai A, Hodosh M, Hockenmaier J (2014) From image descriptions to visual denotations: New similarity metrics for semantic inference over event descriptions. Trans Assoc Comput Linguist 2:67\u201378","journal-title":"Trans Assoc Comput Linguist"},{"key":"5389_CR66","doi-asserted-by":"crossref","unstructured":"Plummer BA, Wang L, Cervantes CM, Caicedo JC, Hockenmaier J, Lazebnik S (2015) Flickr30k entities: Collecting region-to-phrase correspondences for richer image-to-sentence models. In: Proceedings of the IEEE international conference on computer vision, pp 2641\u20132649","DOI":"10.1109\/ICCV.2015.303"},{"key":"5389_CR67","doi-asserted-by":"crossref","unstructured":"Karpathy A, Fei-Fei L (2015) Deep visual-semantic alignments for generating image descriptions. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 3128\u20133137","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"5389_CR68","doi-asserted-by":"crossref","unstructured":"Li Y, Pan Y, Yao T, Mei T (2022) Comprehending and ordering semantics for image captioning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 17990\u201317999","DOI":"10.1109\/CVPR52688.2022.01746"},{"key":"5389_CR69","first-page":"1","volume":"28","author":"S Ren","year":"2015","unstructured":"Ren S, He K, Girshick R, Sun J (2015) Faster r-cnn: Towards real-time object detection with region proposal networks. Adv Neural Inf Process Syst 28:1","journal-title":"Adv Neural Inf Process Syst"},{"key":"5389_CR70","doi-asserted-by":"crossref","unstructured":"Deng J, Dong W, Socher R, Li L-J, Li K, Fei-Fei L (2009) Imagenet: A large-scale hierarchical image database. In: 2009 IEEE conference on computer vision and pattern recognition, pp 248\u2013255. IEEE","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"5389_CR71","unstructured":"Kingma DP, Ba J (2014) Adam: A method for stochastic optimization. arXiv:1412.6980"},{"key":"5389_CR72","doi-asserted-by":"crossref","unstructured":"Papineni K, Roukos S, Ward T, Zhu W-J (2002) Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the 40th annual meeting of the association for computational linguistics, pp 311\u2013318","DOI":"10.3115\/1073083.1073135"},{"key":"5389_CR73","unstructured":"Banerjee S, Lavie A (2005) Meteor: An automatic metric for mt evaluation with improved correlation with human judgments. In: Proceedings of the Acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization, pp 65\u201372"},{"key":"5389_CR74","unstructured":"Lin C-Y (2004) Rouge: A package for automatic evaluation of summaries. In: Text summarization branches out, pp 74\u201381"},{"key":"5389_CR75","doi-asserted-by":"crossref","unstructured":"Vedantam R, Lawrence-Zitnick C, Parikh D (2015) Cider: Consensus-based image description evaluation. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 4566\u20134575","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"5389_CR76","doi-asserted-by":"crossref","unstructured":"Anderson P, Fernando B, Johnson M, Gould S (2016) Spice: Semantic propositional image caption evaluation. In: European conference on computer vision, pp 382\u2013398. Springer","DOI":"10.1007\/978-3-319-46454-1_24"},{"key":"5389_CR77","doi-asserted-by":"publisher","first-page":"117174","DOI":"10.1016\/j.eswa.2022.117174","volume":"201","author":"C Wang","year":"2022","unstructured":"Wang C, Shen Y, Ji L (2022) Geometry attention transformer with position-aware lstms for image captioning. Expert Syst Appl 201:117174","journal-title":"Expert Syst Appl"},{"key":"5389_CR78","doi-asserted-by":"crossref","unstructured":"Biten AF, G\u00f3mez L, Karatzas D (2022) Let there be a clock on the beach: Reducing object hallucination in image captioning. In: 2022 IEEE\/CVF winter conference on applications of computer vision (WACV), pp 2473\u20132482. 10.1109\/WACV51458.2022.00253","DOI":"10.1109\/WACV51458.2022.00253"},{"key":"5389_CR79","first-page":"8320","volume":"33","author":"L Gao","year":"2019","unstructured":"Gao L, Fan K, Song J, Liu X, Xu X, Shen HT (2019) Deliberate attention networks for image captioning. Proc AAAI Conf Artif Intell 33:8320\u20138327","journal-title":"Proc AAAI Conf Artif Intell"},{"issue":"2","key":"5389_CR80","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3439734","volume":"17","author":"H Wei","year":"2021","unstructured":"Wei H, Li Z, Huang F, Zhang C, Ma H, Shi Z (2021) Integrating scene semantic knowledge into image captioning. ACM Trans Multimed Comput Commun Appl (TOMM) 17(2):1\u201322","journal-title":"ACM Trans Multimed Comput Commun Appl (TOMM)"},{"issue":"10","key":"5389_CR81","doi-asserted-by":"publisher","first-page":"7005","DOI":"10.1109\/TCSVT.2022.3178844","volume":"32","author":"S Cao","year":"2022","unstructured":"Cao S, An G, Zheng Z, Wang Z (2022) Vision-enhanced and consensus-aware transformer for image captioning. IEEE Trans Circ Syst Vid Technol 32(10):7005\u20137018","journal-title":"IEEE Trans Circ Syst Vid Technol"},{"key":"5389_CR82","doi-asserted-by":"crossref","unstructured":"Mao Y, Chen L, Jiang Z, Zhang D, Zhang Z, Shao J, Xiao J (2022) Rethinking the reference-based distinctive image captioning. In: Proceedings of the 30th ACM international conference on multimedia, pp 4374\u20134384","DOI":"10.1145\/3503161.3548358"},{"key":"5389_CR83","doi-asserted-by":"publisher","first-page":"109420","DOI":"10.1016\/j.patcog.2023.109420","volume":"138","author":"Y Ma","year":"2023","unstructured":"Ma Y, Ji J, Sun X, Zhou Y, Ji R (2023) Towards local visual modeling for image captioning. Pattern Recognit 138:109420","journal-title":"Pattern Recognit"},{"key":"5389_CR84","doi-asserted-by":"publisher","first-page":"104591","DOI":"10.1016\/j.imavis.2022.104591","volume":"129","author":"Z Li","year":"2023","unstructured":"Li Z, Wei J, Huang F, Ma H (2023) Modeling graph-structured contexts for image captioning. Image Vis Comput 129:104591","journal-title":"Image Vis Comput"},{"key":"5389_CR85","doi-asserted-by":"crossref","unstructured":"Wang C, Gu X (2023) Learning joint relationship attention network for image captioning. Expert Syst Appl 211:118474","DOI":"10.1016\/j.eswa.2022.118474"},{"key":"5389_CR86","doi-asserted-by":"crossref","unstructured":"Kuo C-W, Kira Z (2023) Haav: Hierarchical aggregation of augmented views for image captioning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 11039\u201311049","DOI":"10.1109\/CVPR52729.2023.01062"},{"key":"5389_CR87","doi-asserted-by":"publisher","first-page":"119774","DOI":"10.1016\/j.eswa.2023.119774","volume":"223","author":"H Parvin","year":"2023","unstructured":"Parvin H, Naghsh-Nilchi AR, Mohammadi HM (2023) Transformer-based local-global guidance for image captioning. Expert Syst Appl 223:119774","journal-title":"Expert Syst Appl"},{"key":"5389_CR88","doi-asserted-by":"crossref","unstructured":"Zhou Y, Zhang Y, Hu Z, Wang M (2021) Semi-autoregressive transformer for image captioning. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 3139\u20133143","DOI":"10.1109\/ICCVW54120.2021.00350"},{"key":"5389_CR89","doi-asserted-by":"publisher","first-page":"3101","DOI":"10.1109\/TMM.2021.3093725","volume":"24","author":"Z Zhang","year":"2022","unstructured":"Zhang Z, Wu Q, Wang Y, Chen F (2022) Exploring pairwise relationships adaptively from linguistic context in image captioning. IEEE Trans Multimed 24:3101\u20133113","journal-title":"IEEE Trans Multimed"},{"key":"5389_CR90","doi-asserted-by":"crossref","unstructured":"Hu N, Ming Y, Fan C, Feng F, Lyu B (2022) Tsfnet: Triple-steam image captioning. IEEE Trans Multimed","DOI":"10.1109\/TMM.2022.3215861"},{"key":"5389_CR91","doi-asserted-by":"publisher","first-page":"104575","DOI":"10.1016\/j.imavis.2022.104575","volume":"128","author":"J Hu","year":"2022","unstructured":"Hu J, Yang Y, Yao L, An Y, Pan L (2022) Position-guided transformer for image captioning. Image Vis Comput 128:104575","journal-title":"Image Vis Comput"},{"key":"5389_CR92","doi-asserted-by":"crossref","unstructured":"Wang Y, Xu J, Sun Y (2022) A visual persistence model for image captioning. Neurocomputing 468:48\u201359","DOI":"10.1016\/j.neucom.2021.10.014"},{"key":"5389_CR93","doi-asserted-by":"publisher","first-page":"265","DOI":"10.1016\/j.neucom.2022.07.068","volume":"506","author":"Y Huang","year":"2022","unstructured":"Huang Y, Chen J, Ma H, Ma H, Ouyang W, Yu C (2022) Attribute assisted teacher-critical training strategies for image captioning. Neurocomputing 506:265\u2013276","journal-title":"Neurocomputing"},{"key":"5389_CR94","doi-asserted-by":"publisher","first-page":"812","DOI":"10.1016\/j.ins.2022.12.018","volume":"623","author":"S Dubey","year":"2023","unstructured":"Dubey S, Olimov F, Rafique MA, Kim J, Jeon M (2023) Label-attention transformer with geometrically coherent objects for image captioning. Inf Sci 623:812\u2013831","journal-title":"Inf Sci"},{"key":"5389_CR95","doi-asserted-by":"publisher","first-page":"102377","DOI":"10.1016\/j.displa.2023.102377","volume":"77","author":"L Chen","year":"2023","unstructured":"Chen L, Yang Y, Hu J, Pan L, Zhai H (2023) Relational-convergent transformer for image captioning. Displays 77:102377","journal-title":"Displays"},{"key":"5389_CR96","doi-asserted-by":"crossref","unstructured":"Yao T, Pan Y, Li Y, Mei T (2018) Exploring visual relationship for image captioning. In: Proceedings of the european conference on computer vision (ECCV), pp 684\u2013699","DOI":"10.1007\/978-3-030-01264-9_42"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-024-05389-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-024-05389-y\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-024-05389-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,4,29]],"date-time":"2024-04-29T13:39:43Z","timestamp":1714397983000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-024-05389-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,3]]},"references-count":96,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2024,3]]}},"alternative-id":["5389"],"URL":"https:\/\/doi.org\/10.1007\/s10489-024-05389-y","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"value":"0924-669X","type":"print"},{"value":"1573-7497","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,3]]},"assertion":[{"value":"9 March 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 March 2024","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interest"}}]}}