{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,17]],"date-time":"2026-04-17T06:30:16Z","timestamp":1776407416577,"version":"3.51.2"},"reference-count":34,"publisher":"Springer Science and Business Media LLC","issue":"24","license":[{"start":{"date-parts":[[2019,9,3]],"date-time":"2019-09-03T00:00:00Z","timestamp":1567468800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2019,9,3]],"date-time":"2019-09-03T00:00:00Z","timestamp":1567468800000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"name":"the Fundamental Research Funds for Beijing Universities","award":["110052971803\/037"],"award-info":[{"award-number":["110052971803\/037"]}]},{"name":"Yuyou Talent Support Plan of North China University of Technology","award":["107051360019XN132\/017"],"award-info":[{"award-number":["107051360019XN132\/017"]}]},{"name":"Special Research Foundation of North China University of Technology","award":["PXM2017_014212_000014"],"award-info":[{"award-number":["PXM2017_014212_000014"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"published-print":{"date-parts":[[2019,12]]},"DOI":"10.1007\/s11042-019-08116-9","type":"journal-article","created":{"date-parts":[[2019,9,3]],"date-time":"2019-09-03T01:04:26Z","timestamp":1567472666000},"page":"35329-35350","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":10,"title":["An image caption method based on object detection"],"prefix":"10.1007","volume":"78","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9779-9466","authenticated-orcid":false,"given":"Danyang","family":"Cao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Menggui","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lei","family":"Gao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2019,9,3]]},"reference":[{"key":"8116_CR1","unstructured":"Bahdanau D, Cho K, Bengio Y (2014) neural machine translation by jointly learning to align and translate. Computer Science"},{"issue":"22","key":"8116_CR2","doi-asserted-by":"publisher","first-page":"31163","DOI":"10.1007\/s11042-019-07895-5","volume":"78","author":"Junchi Bin","year":"2019","unstructured":"Bin J, Gardiner B, Liu Z et al (2019) Attention-based multi-modal fusion for improved real estate appraisal: a case study in Los Angeles. Multimed Tools Appl: 1\u201322. doi: https:\/\/doi.org\/10.1007\/s11042-019-07895-5","journal-title":"Multimedia Tools and Applications"},{"key":"8116_CR3","doi-asserted-by":"crossref","unstructured":"Chen L, Zhang H, Xiao J, et al (2017) SCA-CNN: Spatial and Channel-wise Attention in Convolutional Networks for Image Captioning: 6298\u20136306","DOI":"10.1109\/CVPR.2017.667"},{"key":"8116_CR4","doi-asserted-by":"crossref","unstructured":"Cho K, Merrienboer B V, Gulcehre C, et al (2014) Learning phrase representations using RNN encoder-decoder for statistical machine translation. Computer Science","DOI":"10.3115\/v1\/D14-1179"},{"key":"8116_CR5","doi-asserted-by":"crossref","unstructured":"Fang F, Li Q, Wang H, et al (2018) Refining attention: a sequential attention model for image captioning. 2018 IEEE international conference on multimedia and expo (ICME): 1\u20136","DOI":"10.1109\/ICME.2018.8486437"},{"issue":"14","key":"8116_CR6","doi-asserted-by":"publisher","first-page":"20533","DOI":"10.1007\/s11042-019-7404-z","volume":"78","author":"H Ge","year":"2019","unstructured":"Ge H, Yan Z, Yu W et al (2019) An attention mechanism based convolutional LSTM network for video action recognition. Multimed Tools Appl 78(14):20533\u201320556. https:\/\/doi.org\/10.1007\/s11042-019-7404-z","journal-title":"Multimed Tools Appl"},{"key":"8116_CR7","doi-asserted-by":"crossref","unstructured":"Guo Y, Liu Y, De Boer MHT, Liu L, Michael S (2018) A dual prediction network for image captioning, 2018 IEEE international conference on multimedia and expo (ICME): 1\u20136","DOI":"10.1109\/ICME.2018.8486491"},{"issue":"9","key":"8116_CR8","doi-asserted-by":"publisher","first-page":"1904","DOI":"10.1109\/TPAMI.2015.2389824","volume":"37","author":"K He","year":"2015","unstructured":"He K, Zhang X, Ren S et al (2015) Spatial pyramid pooling in deep convolutional networks for visual recognition. IEEE Trans Pattern Anal Mach Intell 37(9):1904\u20131916","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"8116_CR9","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, et al (2015) Deep residual learning for image recognition: 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"8116_CR10","doi-asserted-by":"crossref","unstructured":"Jia X, Gavves E, Fernando B, et al (2015) Guiding long-short term memory for image caption generation","DOI":"10.1109\/ICCV.2015.277"},{"key":"8116_CR11","doi-asserted-by":"crossref","unstructured":"Karpathy A, Li FF (2015) Deep visual-semantic alignments for generating image descriptions. Computer Vision and Pattern Recognition IEEE: 3128\u20133137","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"8116_CR12","unstructured":"Krizhevsky A, Sutskever I, Hinton GE (2012) ImageNet classification with deep convolutional neural networks. International conference on neural information processing systems. Curran Associates Inc: 1097\u20131105"},{"issue":"12","key":"8116_CR13","doi-asserted-by":"publisher","first-page":"2891","DOI":"10.1109\/TPAMI.2012.162","volume":"35","author":"G Kulkarni","year":"2013","unstructured":"Kulkarni G, Premraj V, Ordonez V et al (2013) Babytalk: understanding and generating simple image descriptions. IEEE Trans Pattern Anal Mach Intell 35(12):2891\u20132903","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"8116_CR14","unstructured":"Kuznetsova P, Ordonez V, Berg AC, et al (2012) Collective generation of natural image descriptions. Meeting of the Association for Computational Linguistics: Long Papers Association for Computational Linguistics: 359\u2013368"},{"key":"8116_CR15","unstructured":"Kuznetsova P, Ordonez V, Berg A, et al (2013) Generalizing image captions for image-text parallel Corpus. Meeting of the Association for Computational Linguistics: 790\u2013796"},{"issue":"4","key":"8116_CR16","doi-asserted-by":"publisher","first-page":"541","DOI":"10.1162\/neco.1989.1.4.541","volume":"1","author":"Y Lecun","year":"1989","unstructured":"Lecun Y, Boser B, Denker JS et al (1989) Back propagation applied to handwritten zip code recognition. Neural Comput 1(4):541\u2013551","journal-title":"Neural Comput"},{"key":"8116_CR17","unstructured":"Lipton Z C, Berkowitz J, Elkan C (2015) A critical review of recurrent neural networks for sequence learning. Computer Science"},{"key":"8116_CR18","doi-asserted-by":"crossref","unstructured":"Lu J, Xiong C, Parikh D, et al (2016) Knowing when to look: adaptive attention via a visual sentinel for image captioning: 3242\u20133250","DOI":"10.1109\/CVPR.2017.345"},{"key":"8116_CR19","unstructured":"Mitchell M, Han X, Dodge J, et al (2012) Midge: generating image descriptions from computer vision detections. Conference of the European chapter of the Association for Computational Linguistics. Association for Computational Linguistics: 747\u2013756"},{"key":"8116_CR20","unstructured":"Ren S, He K, Girshick R, et al (2015) Faster R-CNN: towards real-time object detection with region proposal networks. International conference on neural information processing systems. MIT Press: 91\u201399"},{"key":"8116_CR21","unstructured":"Sadeghi MA, Sadeghi MA, Sadeghi MA, et al (2010) Every picture tells a story: generating sentences from images. European conference on computer vision. Springer-Verlag: 15\u201329"},{"key":"8116_CR22","unstructured":"Sak H, Senior A, Beaufays F (2014) Long short-term memory based recurrent neural network architectures for large vocabulary speech recognition. Computer Science: 338\u2013342"},{"key":"8116_CR23","unstructured":"Simonyan K, Zisserman A (2014) Very deep convolutional networks for large-scale image recognition. Computer Science"},{"key":"8116_CR24","doi-asserted-by":"crossref","unstructured":"Sundermeyer M, Schl\u00fcter R, Ney H (2012) LSTM neural networks for language modeling. Interspeech: 601\u2013608","DOI":"10.21437\/Interspeech.2012-65"},{"key":"8116_CR25","doi-asserted-by":"crossref","unstructured":"Szegedy C, Liu W, Jia Y, et al (2014) Going deeper with convolutions: 1\u20139","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"8116_CR26","doi-asserted-by":"crossref","unstructured":"Vinyals O, Toshev A, Bengio S, et al (2014) Show and tell: a neural image caption generator: 3156\u20133164","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"8116_CR27","doi-asserted-by":"crossref","unstructured":"Wu Q, Shen C, Liu L, et al (2016) What value do explicit high level concepts have in vision to language problems?, Computer Science: 203\u2013212","DOI":"10.1109\/CVPR.2016.29"},{"key":"8116_CR28","unstructured":"Xu K, Ba J, Kiros R, et al (2015) Show, attend and tell: neural image caption generation with visual attention. Computer Science: 2048\u20132057"},{"key":"8116_CR29","unstructured":"Yang Y, Teo CL, Aloimonos Y (2011) Corpus-guided sentence generation of natural images. Conference on empirical methods in natural language processing. Association for Computational Linguistics: 444\u2013454"},{"key":"8116_CR30","unstructured":"Yang Z, Yuan Y, Wu Y, et al (2016) Encode, review, and decode: reviewer module for caption generation"},{"key":"8116_CR31","doi-asserted-by":"crossref","unstructured":"Yao T, Pan Y, Li Y, et al (2016) Boosting image captioning with attributes: 4904\u20134912","DOI":"10.1109\/ICCV.2017.524"},{"issue":"11","key":"8116_CR32","doi-asserted-by":"publisher","first-page":"5514","DOI":"10.1109\/TIP.2018.2855406","volume":"27","author":"Senmao Ye","year":"2018","unstructured":"Ye S, Liu N, Han J (2018) Attentive linear transformation for image captioning. IEEE Trans Image Process: 5514\u20135524","journal-title":"IEEE Transactions on Image Processing"},{"key":"8116_CR33","doi-asserted-by":"crossref","unstructured":"Zhou Y, Zhenzhen H, Ye Z, Liu X, Hong R (2018) Enhanced text-guided attention model for image captioning. 2018 IEEE fourth international conference on multimedia big data (BigMM): 1\u20135","DOI":"10.1109\/BigMM.2018.8499172"},{"key":"8116_CR34","doi-asserted-by":"crossref","unstructured":"Zhu Z, Xue Z, Yuan Z (2018) Topic-guided attention for image captioning: 2615\u20132619","DOI":"10.1109\/ICIP.2018.8451083"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-019-08116-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11042-019-08116-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-019-08116-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,9,27]],"date-time":"2022-09-27T05:52:24Z","timestamp":1664257944000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11042-019-08116-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,9,3]]},"references-count":34,"journal-issue":{"issue":"24","published-print":{"date-parts":[[2019,12]]}},"alternative-id":["8116"],"URL":"https:\/\/doi.org\/10.1007\/s11042-019-08116-9","relation":{},"ISSN":["1380-7501","1573-7721"],"issn-type":[{"value":"1380-7501","type":"print"},{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019,9,3]]},"assertion":[{"value":"13 September 2018","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 July 2019","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 August 2019","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 September 2019","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}