{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,8]],"date-time":"2026-05-08T12:37:56Z","timestamp":1778243876925,"version":"3.51.4"},"reference-count":45,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2022,3,23]],"date-time":"2022-03-23T00:00:00Z","timestamp":1647993600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,3,23]],"date-time":"2022-03-23T00:00:00Z","timestamp":1647993600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61571453"],"award-info":[{"award-number":["61571453"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61806218"],"award-info":[{"award-number":["61806218"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Multimed Info Retr"],"published-print":{"date-parts":[[2022,6]]},"DOI":"10.1007\/s13735-022-00228-7","type":"journal-article","created":{"date-parts":[[2022,3,23]],"date-time":"2022-03-23T05:02:30Z","timestamp":1648011750000},"page":"111-121","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":11,"title":["Caption TLSTMs: combining transformer with LSTMs for image captioning"],"prefix":"10.1007","volume":"11","author":[{"given":"Jie","family":"Yan","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6079-6907","authenticated-orcid":false,"given":"Yuxiang","family":"Xie","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xidao","family":"Luan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yanming","family":"Guo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Quanzhi","family":"Gong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Suru","family":"Feng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,3,23]]},"reference":[{"key":"228_CR1","unstructured":"Kiros R, Salakhutdinov R, Zemel R (2014) Multimodal neural language models, In: International conference on machine learning, pp 595-603, PMLR"},{"issue":"6","key":"228_CR2","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3295748","volume":"51","author":"MZ Hossain","year":"2019","unstructured":"Hossain MZ, Sohel F, Shiratuddin MF, Laga HJACS (2019) A comprehensive survey of deep learning for image captioning. ACM Comput Surv 51(6):1\u201336","journal-title":"ACM Comput Surv"},{"key":"228_CR3","doi-asserted-by":"publisher","first-page":"92","DOI":"10.1016\/j.neucom.2020.02.041","volume":"396","author":"R Li","year":"2020","unstructured":"Li R, Liang H, Shi Y, Feng F, Wang XJN (2020) Dual-CNN: a convolutional language decoder for paragraph image captioning. Neurocomputing 396:92\u2013101","journal-title":"Neurocomputing"},{"key":"228_CR4","doi-asserted-by":"crossref","unstructured":"Papineni K, Roukos S, Ward T, W-J. Zhu (2002) Bleu: a method for automatic evaluation of machine translation, In: Proceedings of the 40th annual meeting of the association for computational linguistics, pp 311-318","DOI":"10.3115\/1073083.1073135"},{"key":"228_CR5","unstructured":"Banerjee S, Lavie A (2005) METEOR: An automatic metric for MT evaluation with improved correlation with human judgments, In: Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization, pp 65-72"},{"key":"228_CR6","unstructured":"Lin C-Y (2004) Rouge: a package for automatic evaluation of summaries, In: Text summarization branches out, pp 74-81"},{"key":"228_CR7","doi-asserted-by":"crossref","unstructured":"Vedantam R, Lawrence Zitnick C, Parikh D (2015) Cider: consensus-based image description evaluation, In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 4566-4575","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"228_CR8","doi-asserted-by":"crossref","unstructured":"Chen S, Jin Q, Wang P, Wu Q (2020) Say as you wish: Fine-grained control of image caption generation with abstract scene graphs, In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 9962-9971","DOI":"10.1109\/CVPR42600.2020.00998"},{"key":"228_CR9","doi-asserted-by":"crossref","unstructured":"Vinyals O, Toshev A, Bengio S, Erhan D (2015) Show and tell: a neural image caption generator, In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 3156-3164","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"228_CR10","doi-asserted-by":"crossref","unstructured":"Anderson P et al (2018) Bottom-Up and Top-Down Attention for Image Captioning and Visual Question Answering, In: 31st meeting of the IEEE\/CVF conference on computer vision and pattern recognition, CVPR 2018, June 18, 2018 - June 22, 2018, Salt Lake City, UT, United states, pp 6077-6086: IEEE Computer Society","DOI":"10.1109\/CVPR.2018.00636"},{"key":"228_CR11","doi-asserted-by":"crossref","unstructured":"Kulkarni G et al. (2011) Baby talk: understanding and generating simple image descriptions, In: 2011 IEEE conference on computer vision and pattern recognition, CVPR 2011, pp 1601-1608: IEEE Computer Society","DOI":"10.1109\/CVPR.2011.5995466"},{"key":"228_CR12","doi-asserted-by":"crossref","unstructured":"Yang X, Liu Y, Wang XJ (2021) ReFormer: the relational transformer for image captioning","DOI":"10.1109\/RCAE53607.2021.9638904"},{"key":"228_CR13","doi-asserted-by":"crossref","unstructured":"Jiang W, Ma L, Jiang Y-G, Liu W, Zhang T (2018) Recurrent fusion network for image captioning, In: Proceedings of the european conference on computer vision (ECCV), pp 499-515","DOI":"10.1007\/978-3-030-01216-8_31"},{"key":"228_CR14","unstructured":"Xu K et al (2015) Show, attend and tell: Neural image caption generation with visual attention, In: 32nd international conference on machine learning, ICML 2015, July 6, 2015 - July 11, 2015, Lile, France, vol. 3, pp. 2048-2057: International machine learning society (IMLS)"},{"key":"228_CR15","doi-asserted-by":"crossref","unstructured":"Xu D, Zhu Y, Choy CB, Fei-Fei L (2017) Scene graph generation by iterative message passing, in Proceedings of the IEEE conference on computer vision and pattern recognition, pp 5410-5419","DOI":"10.1109\/CVPR.2017.330"},{"key":"228_CR16","doi-asserted-by":"crossref","unstructured":"Zellers R, Yatskar M, Thomson S, Choi Y (2018) Neural motifs: Scene graph parsing with global context, In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 5831-5840","DOI":"10.1109\/CVPR.2018.00611"},{"key":"228_CR17","doi-asserted-by":"crossref","unstructured":"Sun G, Zhang C, Woodland PC (2021) Transformer language models with lstm-based cross-utterance information representation, In: ICASSP 2021-2021 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp. 7363-7367: IEEE","DOI":"10.1109\/ICASSP39728.2021.9414477"},{"key":"228_CR18","doi-asserted-by":"crossref","unstructured":"Graves A, Jaitly N, Mohamed A-r (2013) Hybrid speech recognition with deep bidirectional LSTM, In: 2013 IEEE workshop on automatic speech recognition and understanding, 2013, pp. 273-278: IEEE","DOI":"10.1109\/ASRU.2013.6707742"},{"key":"228_CR19","doi-asserted-by":"crossref","unstructured":"Soltau H, Liao H, Sak HJ (2016) Neural speech recognizer: Acoustic-to-word LSTM model for large vocabulary speech recognition","DOI":"10.21437\/Interspeech.2017-1566"},{"key":"228_CR20","doi-asserted-by":"crossref","unstructured":"Dai Z, Yang Z, Yang Y, Carbonell J, Le QV, Salakhutdinov RJ (2019) Transformer-xl: attentive language models beyond a fixed-length context. arXiv preprint, arXiv:1901.02860","DOI":"10.18653\/v1\/P19-1285"},{"key":"228_CR21","doi-asserted-by":"crossref","unstructured":"Irie K, Zeyer A, Schl\u00fcter R, Ney HJ (2019) Language modeling with deep transformers","DOI":"10.21437\/Interspeech.2019-2225"},{"key":"228_CR22","unstructured":"Vaswani A et al (2017) Attention is all you need. Advances in neural information processing systems 30, pp 5998-6008"},{"key":"228_CR23","doi-asserted-by":"crossref","unstructured":"Farhadi A et al (2010) Every picture tells a story: Generating sentences from images, In: European conference on computer vision, pp 15-29, Springer","DOI":"10.1007\/978-3-642-15561-1_2"},{"key":"228_CR24","doi-asserted-by":"crossref","unstructured":"Gupta A, Verma Y, Jawahar C (2012) Choosing linguistics over vision to describe images, In: Proceedings of the AAAI conference on artificial intelligence, vol. 26(1)","DOI":"10.1609\/aaai.v26i1.8205"},{"key":"228_CR25","unstructured":"Ordonez V, Kulkarni G, Berg TJA (2011) Im2text: describing images using 1 million captioned photographs, vol. 24, pp 1143-1151"},{"key":"228_CR26","doi-asserted-by":"crossref","unstructured":"Zhang X et al. (2021) RSTNet: Captioning With Adaptive Attention on Visual and Non-Visual Words, In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 15465-15474","DOI":"10.1109\/CVPR46437.2021.01521"},{"key":"228_CR27","doi-asserted-by":"crossref","unstructured":"Ushiku Y, Yamaguchi M, Mukuta Y, Harada T (2015) Common subspace for model and similarity: Phrase learning for caption generation from images, In: Proceedings of the IEEE international conference on computer vision, pp 2668-2676","DOI":"10.1109\/ICCV.2015.306"},{"key":"228_CR28","unstructured":"Mitchell M et al. (2012) Midge: Generating image descriptions from computer vision detections, In: Proceedings of the 13th conference of the european chapter of the association for computational linguistics, pp 747-756"},{"key":"228_CR29","unstructured":"Sutskever I, Vinyals O, Le QVJ (2014) Sequence to sequence learning with neural networks"},{"key":"228_CR30","doi-asserted-by":"crossref","unstructured":"Jia X, Gavves E, Fernando B, Tuytelaars T (2015) Guiding the long-short term memory model for image caption generation, In: 15th IEEE international conference on computer vision, ICCV 2015, December 11, 2015 - December 18, 2015, Santiago, Chile, 2015, vol. 2015 international conference on computer vision, ICCV 2015, pp. 2407-2415: Institute of Electrical and Electronics Engineers Inc","DOI":"10.1109\/ICCV.2015.277"},{"issue":"5","key":"228_CR31","doi-asserted-by":"publisher","first-page":"739","DOI":"10.3390\/app8050739","volume":"8","author":"X Zhu","year":"2018","unstructured":"Zhu X, Li L, Liu J, Peng H, Niu XJAS (2018) Captioning transformer with stacked attention modules. Appl Sci 8(5):739","journal-title":"Appl Sci"},{"key":"228_CR32","doi-asserted-by":"crossref","unstructured":"Johnson J et al. (2015) Image retrieval using scene graphs, In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 3668-3678","DOI":"10.1109\/CVPR.2015.7298990"},{"key":"228_CR33","unstructured":"Han K et al (2020) A survey on visual transformer. arXiv e-prints, arXiv:2111.06091"},{"key":"228_CR34","doi-asserted-by":"crossref","unstructured":"Khan S, Naseer M, Hayat M, Zamir SW, Khan FS, Shah MJ (2021) Transformers in vision: a survey. ACM Computing Surveys (CSUR)","DOI":"10.1145\/3505244"},{"key":"228_CR35","unstructured":"Herdade S, Kappeler A, Boakye K, Soares JJ (2019) Image captioning: transforming objects into words"},{"key":"228_CR36","doi-asserted-by":"crossref","unstructured":"Li G, Zhu L, Liu P, Yang Y (2019) Entangled transformer for image captioning, In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 8928-8937","DOI":"10.1109\/ICCV.2019.00902"},{"key":"228_CR37","doi-asserted-by":"crossref","unstructured":"Cornia M, Stefanini M, Baraldi L, Cucchiara R (2020) Meshed-memory transformer for image captioning, In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 10578-10587","DOI":"10.1109\/CVPR42600.2020.01059"},{"key":"228_CR38","doi-asserted-by":"crossref","unstructured":"Pan Y, Yao T, Li Y, Mei T (2020) X-linear attention networks for image captioning, In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 10971-10980","DOI":"10.1109\/CVPR42600.2020.01098"},{"key":"228_CR39","unstructured":"Liu X, Zhang P, Yu C, Lu H, Qian X, Yang XJ (2021) A video is worth three views: trigeminal transformers for video-based person re-identification"},{"key":"228_CR40","doi-asserted-by":"crossref","unstructured":"Schlichtkrull M, Kipf TN, Bloem P, Van Den Berg R, Titov I, Welling M (2018) Modeling relational data with graph convolutional networks, In: European semantic web conference, pp 593-607, Springer","DOI":"10.1007\/978-3-319-93417-4_38"},{"issue":"1","key":"228_CR41","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1007\/s11263-016-0981-7","volume":"123","author":"R Krishna","year":"2017","unstructured":"Krishna R et al (2017) Visual genome: connecting language and vision using crowdsourced dense image annotations. Int J Comput Vis 123(1):32\u201373","journal-title":"Int J Comput Vis"},{"key":"228_CR42","doi-asserted-by":"crossref","unstructured":"Lin T-Y et al (2014) Microsoft coco: Common objects in context, In: European conference on computer vision, pp 740-755, Springer","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"228_CR43","doi-asserted-by":"crossref","unstructured":"Karpathy A, Fei-Fei L (2015) Deep visual-semantic alignments for generating image descriptions, In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 3128-3137","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"228_CR44","first-page":"91","volume":"28","author":"S Ren","year":"2015","unstructured":"Ren S, He K, Girshick R, Sun JJA (2015) Faster r-cnn: towards real-time object detection with region proposal networks. Adv Neural Inf Process Syst 28:91-99","journal-title":"Adv Neural Inf Process Syst"},{"key":"228_CR45","doi-asserted-by":"crossref","unstructured":"Deng J, Dong W, Socher R, Li L-J, Li K, Fei-Fei L (2009) Imagenet: a large-scale hierarchical image database, In: 2009 IEEE conference on computer vision and pattern recognition, pp 248-255: Ieee","DOI":"10.1109\/CVPR.2009.5206848"}],"container-title":["International Journal of Multimedia Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-022-00228-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13735-022-00228-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-022-00228-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,1,29]],"date-time":"2023-01-29T20:30:04Z","timestamp":1675024204000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13735-022-00228-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,3,23]]},"references-count":45,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2022,6]]}},"alternative-id":["228"],"URL":"https:\/\/doi.org\/10.1007\/s13735-022-00228-7","relation":{},"ISSN":["2192-6611","2192-662X"],"issn-type":[{"value":"2192-6611","type":"print"},{"value":"2192-662X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,3,23]]},"assertion":[{"value":"7 December 2021","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 February 2022","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 March 2022","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"23 March 2022","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}