{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,25]],"date-time":"2026-08-25T22:54:08Z","timestamp":1787698448802,"version":"build-2784847793"},"publisher-location":"New York, NY, USA","reference-count":33,"publisher":"ACM","license":[{"start":{"date-parts":[[2018,12,21]],"date-time":"2018-12-21T00:00:00Z","timestamp":1545350400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2018,12,21]]},"DOI":"10.1145\/3302425.3302464","type":"proceedings-article","created":{"date-parts":[[2019,2,6]],"date-time":"2019-02-06T19:17:35Z","timestamp":1549480655000},"page":"1-6","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":6,"title":["AttResNet"],"prefix":"10.1145","author":[{"given":"Yunmeng","family":"Feng","sequence":"first","affiliation":[{"name":"Science and Technology on Parallel and Distributed Laboratory, College of Computer, National University of Defense Technology, Changsha, Hunan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Long","family":"Lan","sequence":"additional","affiliation":[{"name":"Institute for Quantum Information &amp; State Key Laboratory of High Performance Computing, College of Computer, National University of Defense Technology, Changsha, Hunan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiang","family":"Zhang","sequence":"additional","affiliation":[{"name":"Institute for Quantum Information &amp; State Key Laboratory of High Performance Computing, College of Computer, National University of Defense Technology, Changsha, Hunan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chuanfu","family":"Xu","sequence":"additional","affiliation":[{"name":"Institute for Quantum Information &amp; State Key Laboratory of High Performance Computing, College of Computer, National University of Defense Technology, Changsha, Hunan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhenghua","family":"Wang","sequence":"additional","affiliation":[{"name":"Science and Technology on Parallel and Distributed Laboratory, College of Computer, National University of Defense Technology, Changsha, Hunan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhigang","family":"Luo","sequence":"additional","affiliation":[{"name":"Science and Technology on Parallel and Distributed Laboratory, College of Computer, National University of Defense Technology, Changsha, Hunan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2018,12,21]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"crossref","unstructured":"Fang H Gupta S Iandola F Srivastava R. K Deng L Dollar P Gao J He X Mitchell M Platt J. C et al(2015). From captions to visual concepts and back. Computer Vision and Pattern Recognition 1473--1482.  Fang H Gupta S Iandola F Srivastava R. K Deng L Dollar P Gao J He X Mitchell M Platt J. C et al(2015). From captions to visual concepts and back. Computer Vision and Pattern Recognition 1473--1482.","DOI":"10.1109\/CVPR.2015.7298754"},{"key":"e_1_3_2_1_2_1","volume-title":"R et al(2015). Show, Attend and Tell: Neural Image Caption Generation with Visual Attention","author":"Xu","year":"2048","unstructured":"Xu K Ba, J. Kiros . R et al(2015). Show, Attend and Tell: Neural Image Caption Generation with Visual Attention , 2048 --2057. Xu K Ba, J. Kiros. R et al(2015). Show, Attend and Tell: Neural Image Caption Generation with Visual Attention, 2048--2057."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","unstructured":"Vinyals O Toshev A Bengio S Erhan D(2015). Show and tell: A neural image caption generator. Computer Vision and Pattern Recognition 3156--3164.  Vinyals O Toshev A Bengio S Erhan D(2015). Show and tell: A neural image caption generator. Computer Vision and Pattern Recognition 3156--3164.","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"crossref","unstructured":"Karpathy A Fei-Fei L(2015). Deep visual-semantic alignments for generating image descriptions. Computer Vision and Pattern Recognition 3128--3137.  Karpathy A Fei-Fei L(2015). Deep visual-semantic alignments for generating image descriptions. Computer Vision and Pattern Recognition 3128--3137.","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"e_1_3_2_1_5_1","unstructured":"Mao J Xu W Yang Y et al(2014). Deep captioning with multimodal recurrent neural networks (m-rnn). arXiv preprint arXiv 1412.6632.  Mao J Xu W Yang Y et al(2014). Deep captioning with multimodal recurrent neural networks (m-rnn). arXiv preprint arXiv 1412.6632."},{"key":"e_1_3_2_1_6_1","unstructured":"Yang Z Yuan Y Wu Y etal: Encode Review and Decode: Reviewer Module for Caption Generation. arXiv preprint arXiv:1605.07912.   Yang Z Yuan Y Wu Y et al.: Encode Review and Decode: Reviewer Module for Caption Generation. arXiv preprint arXiv:1605.07912."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"crossref","unstructured":"Gan Z Gan C He X et al(2017). Semantic Compositional Networks for Visual Captioning. Computer Vision and Pattern Recognition 1141--1150.  Gan Z Gan C He X et al(2017). Semantic Compositional Networks for Visual Captioning. Computer Vision and Pattern Recognition 1141--1150.","DOI":"10.1109\/CVPR.2017.127"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"crossref","unstructured":"Lu J Xiong C Parikh D et al(2016). Knowing When to Look: Adaptive Attention via A Visual Sentinel for Image Captioning. Computer Vision and Pattern Recognition 3242--3250.  Lu J Xiong C Parikh D et al(2016). Knowing When to Look: Adaptive Attention via A Visual Sentinel for Image Captioning. Computer Vision and Pattern Recognition 3242--3250.","DOI":"10.1109\/CVPR.2017.345"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"crossref","unstructured":"You Q Jin H Wang Z et al(2016). Image Captioning with Semantic Attention. Computer Vision and Pattern Recognition 4651--4659.  You Q Jin H Wang Z et al(2016). Image Captioning with Semantic Attention. Computer Vision and Pattern Recognition 4651--4659.","DOI":"10.1109\/CVPR.2016.503"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"crossref","unstructured":"Chen L Zhang H Xiao J et al(2017). SCA-CNN: Spatial and Channel-wise Attention in Convolutional Networks for Image Captioning. Computer Vision and Pattern Recognition 6298--6306.  Chen L Zhang H Xiao J et al(2017). SCA-CNN: Spatial and Channel-wise Attention in Convolutional Networks for Image Captioning. Computer Vision and Pattern Recognition 6298--6306.","DOI":"10.1109\/CVPR.2017.667"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"crossref","unstructured":"Mun J Cho M Han B(2017). Text-Guided Attention Model for Image Captioning. AAAI 4233--4239.   Mun J Cho M Han B(2017). Text-Guided Attention Model for Image Captioning. AAAI 4233--4239.","DOI":"10.1609\/aaai.v31i1.11237"},{"key":"e_1_3_2_1_12_1","unstructured":"Venugopalan S Hendricks L A Rohrbach M et al(2016). Captioning images with diverse objects. arXiv preprint arXiv 1(3) 1606.07770.  Venugopalan S Hendricks L A Rohrbach M et al(2016). Captioning images with diverse objects. arXiv preprint arXiv 1(3) 1606.07770."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"crossref","unstructured":"Chen M Ding G Zhao S etal(2017). Reference Based LSTM for Image Captioning. AAAI 3981--3987.   Chen M Ding G Zhao S et al(2017). Reference Based LSTM for Image Captioning. AAAI 3981--3987.","DOI":"10.1609\/aaai.v31i1.11198"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/2964284.2964299"},{"key":"e_1_3_2_1_15_1","volume-title":"What value do explicit high level concepts have in vision to language problems? Computer Vision and Pattern Recognition, 203--212","author":"Wu Q","year":"2016","unstructured":"Wu Q , Shen C , Liu L ( 2016 ). What value do explicit high level concepts have in vision to language problems? Computer Vision and Pattern Recognition, 203--212 . Wu Q, Shen C, Liu L et al (2016). What value do explicit high level concepts have in vision to language problems? Computer Vision and Pattern Recognition, 203--212."},{"key":"e_1_3_2_1_16_1","unstructured":"D Bahdanau K Cho Y Bengio(2014). Neural machine translation by jointly learning to align and translate. arXiv preprint arXiv 1409.0473.  D Bahdanau K Cho Y Bengio(2014). Neural machine translation by jointly learning to align and translate. arXiv preprint arXiv 1409.0473."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"crossref","unstructured":"Yang Z He X Gao J et al(2016). Stacked attention networks for image question answering. Computer Vision and Pattern Recognition 21--29.  Yang Z He X Gao J et al(2016). Stacked attention networks for image question answering. Computer Vision and Pattern Recognition 21--29.","DOI":"10.1109\/CVPR.2016.10"},{"key":"e_1_3_2_1_18_1","series-title":"Lecture Notes in Computer Science, 15--29","volume-title":"Every picture tells a story: Generating sentences from images. European conference on computer vision","author":"Farhadi A","unstructured":"Farhadi A , Hejrati M , Sadeghi M A , Every picture tells a story: Generating sentences from images. European conference on computer vision . Lecture Notes in Computer Science, 15--29 . Farhadi A, Hejrati M, Sadeghi M A, et al(2010). Every picture tells a story: Generating sentences from images. European conference on computer vision. Lecture Notes in Computer Science, 15--29."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2012.162"},{"key":"e_1_3_2_1_20_1","volume-title":"Proceedings of the 50th Annual Meeting of the Association for Computational Linguistics, 1, 359--368","author":"Kuznetsova P","unstructured":"Kuznetsova P , Ordonez V , Berg A. C Collective generation of natural image descriptions . Proceedings of the 50th Annual Meeting of the Association for Computational Linguistics, 1, 359--368 . Kuznetsova P, Ordonez V, Berg A. C et al(2012). Collective generation of natural image descriptions. Proceedings of the 50th Annual Meeting of the Association for Computational Linguistics, 1, 359--368."},{"key":"e_1_3_2_1_21_1","volume-title":"Proceedings of the 13th Conference of the European Chapter of the Association for Computational Linguistics, 747--756","author":"Mitchell M","unstructured":"Mitchell M , Han X , Dodge J Midge : Generating image descriptions from computer vision detections . Proceedings of the 13th Conference of the European Chapter of the Association for Computational Linguistics, 747--756 . Mitchell M, Han X, Dodge J et al(2012). Midge: Generating image descriptions from computer vision detections. Proceedings of the 13th Conference of the European Chapter of the Association for Computational Linguistics, 747--756."},{"key":"e_1_3_2_1_22_1","volume-title":"Proceedings of the 52nd Annual Meeting of the Association for Computational Linguistics 2, 592--598","author":"Mason R","unstructured":"Mason R , Charniak E(2014). Nonparametric method for data-driven image captioning . Proceedings of the 52nd Annual Meeting of the Association for Computational Linguistics 2, 592--598 . Mason R, Charniak E(2014). Nonparametric method for data-driven image captioning. Proceedings of the 52nd Annual Meeting of the Association for Computational Linguistics 2, 592--598."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"crossref","unstructured":"Wang F Jiang M Qian C et al(2017). Residual attention network for image classification. arXiv preprint arXiv 1704.06904.  Wang F Jiang M Qian C et al(2017). Residual attention network for image classification. arXiv preprint arXiv 1704.06904.","DOI":"10.1109\/CVPR.2017.683"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"crossref","unstructured":"K Cho B Van. Merrienboer C Gulcehre D Bahdanau F Bougares H Schwenk Y Bengio(2014). Learning phrase representations using rnn encoder-decoder for statistical machine translation. arXiv preprint arXiv 1406.1078.  K Cho B Van. Merrienboer C Gulcehre D Bahdanau F Bougares H Schwenk Y Bengio(2014). Learning phrase representations using rnn encoder-decoder for statistical machine translation. arXiv preprint arXiv 1406.1078.","DOI":"10.3115\/v1\/D14-1179"},{"key":"e_1_3_2_1_25_1","unstructured":"Sutskever I Vinyals O Le Q V(2014). Sequence to sequence learning with neural networks. Advances in neural information processing systems 3104--3112.   Sutskever I Vinyals O Le Q V(2014). Sequence to sequence learning with neural networks. Advances in neural information processing systems 3104--3112."},{"key":"e_1_3_2_1_26_1","volume-title":"European conference on computer vision, 740--755","author":"Lin T Y","unstructured":"Lin T Y , Maire M , Belongie S Microsoft coco : Common objects in context . European conference on computer vision, 740--755 . Lin T Y, Maire M, Belongie S et al(2014). Microsoft coco: Common objects in context. European conference on computer vision, 740--755."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.3115\/1073083.1073135"},{"key":"e_1_3_2_1_28_1","volume-title":"Proceedings of the ninth workshop on statistical machine translation, 376--380","author":"Denkowski M","unstructured":"Denkowski M , Lavie A(2014). Meteor universal : Language specific translation evaluation for any target language . Proceedings of the ninth workshop on statistical machine translation, 376--380 . Denkowski M, Lavie A(2014). Meteor universal: Language specific translation evaluation for any target language. Proceedings of the ninth workshop on statistical machine translation, 376--380."},{"key":"e_1_3_2_1_29_1","unstructured":"Lin C Y(2004). Rouge: A package for automatic evaluation of summaries. Text Summarization Branches Out.  Lin C Y(2004). Rouge: A package for automatic evaluation of summaries. Text Summarization Branches Out."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"crossref","unstructured":"Vedantam R Lawrence Zitnick C Parikh D(2015). Cider: Consensus-based image description evaluation. Computer Vision and Pattern Recognition 3156--3164.  Vedantam R Lawrence Zitnick C Parikh D(2015). Cider: Consensus-based image description evaluation. Computer Vision and Pattern Recognition 3156--3164.","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"crossref","unstructured":"J Donahue L A. Hendricks S Guadarrama M Rohrbach S Venugopalan K Saenko T Darrell(2015). Long-term recurrent convolutional networks for visual recognition and description. Computer Vision and Pattern Recognition 2626--2634.  J Donahue L A. Hendricks S Guadarrama M Rohrbach S Venugopalan K Saenko T Darrell(2015). Long-term recurrent convolutional networks for visual recognition and description. Computer Vision and Pattern Recognition 2626--2634.","DOI":"10.1109\/CVPR.2015.7298878"},{"key":"e_1_3_2_1_32_1","volume-title":"Deng L et al(2017). Image Caption with Global-Local Attention","author":"Li L","unstructured":"Li L , Tang S , Deng L et al(2017). Image Caption with Global-Local Attention . AAAI. AAAI Press , 4133--4139. Li L, Tang S, Deng L et al(2017). Image Caption with Global-Local Attention. AAAI. AAAI Press, 4133--4139."},{"key":"e_1_3_2_1_33_1","unstructured":"Zhou L Xu C Koch P et al(2016). Image Caption Generation with Text-Conditional Semantic Attention. arXiv preprint arXiv 1606.04621.  Zhou L Xu C Koch P et al(2016). Image Caption Generation with Text-Conditional Semantic Attention. arXiv preprint arXiv 1606.04621."}],"event":{"name":"ACAI 2018: 2018 International Conference on Algorithms, Computing and Artificial Intelligence","location":"Sanya China","acronym":"ACAI 2018","sponsor":["The Hong Kong Polytechnic The Hong Kong Polytechnic University","City University of Hong Kong City University of Hong Kong"]},"container-title":["Proceedings of the 2018 International Conference on Algorithms, Computing and Artificial Intelligence"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3302425.3302464","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3302425.3302464","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T19:05:45Z","timestamp":1750273545000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3302425.3302464"}},"subtitle":["Attention-based ResNet for Image Captioning"],"short-title":[],"issued":{"date-parts":[[2018,12,21]]},"references-count":33,"alternative-id":["10.1145\/3302425.3302464","10.1145\/3302425"],"URL":"https:\/\/doi.org\/10.1145\/3302425.3302464","relation":{},"subject":[],"published":{"date-parts":[[2018,12,21]]},"assertion":[{"value":"2018-12-21","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}