{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:16:32Z","timestamp":1750220192438,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":25,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,9,23]],"date-time":"2022-09-23T00:00:00Z","timestamp":1663891200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,9,23]]},"DOI":"10.1145\/3573942.3574052","type":"proceedings-article","created":{"date-parts":[[2023,5,16]],"date-time":"2023-05-16T23:45:42Z","timestamp":1684280742000},"page":"482-488","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Image Captioning Method Based on Layer Feature Attention"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8231-8562","authenticated-orcid":false,"given":"Qiujuan","family":"Tong","sequence":"first","affiliation":[{"name":"School of Science, Xi'an University of Posts and Telecommunications, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3071-7494","authenticated-orcid":false,"given":"Chan","family":"He","sequence":"additional","affiliation":[{"name":"School of Telecommunication and Information Engineering, Xi'an University of Posts and Telecommunications, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8015-8291","authenticated-orcid":false,"given":"Jiaqi","family":"Li","sequence":"additional","affiliation":[{"name":"School of Science, Xi'an University of Posts and Telecommunications, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5957-7339","authenticated-orcid":false,"given":"Yifan","family":"Li","sequence":"additional","affiliation":[{"name":"School of Science, Xi'an University of Posts and Telecommunications, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,5,16]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Mori Y Takahashi H Oka R. Image-to-word transformation based on dividing and vector quantizing images with words[C]\/\/First international workshop on multimedia intelligent storage and retrieval management. 1999: 1-9."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"crossref","unstructured":"Vinyals O Toshev A Bengio S Show and tell: A neural image caption generator[C]\/\/Proceedings of the IEEE conference on computer vision and pattern recognition. 2015: 3156-3164.","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","unstructured":"Chen S Jiang Y G. Towards bridging event captioner and sentence localizer for weakly supervised dense event captioning[C]\/\/Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 2021: 8425-8435.","DOI":"10.1109\/CVPR46437.2021.00832"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"crossref","unstructured":"Hosseinzadeh M Wang Y. Image Change Captioning by Learning from an Auxiliary Task[C]\/\/Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 2021: 2725-2734.","DOI":"10.1109\/CVPR46437.2021.00275"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"crossref","unstructured":"Chen L Jiang Z Xiao J Human-like controllable image captioning with verb-specific semantic roles[C]\/\/Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 2021: 16846-16856.","DOI":"10.1109\/CVPR46437.2021.01657"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"crossref","unstructured":"Zhang X Sun X Luo Y RSTNet: Captioning with Adaptive Attention on Visual and Non-Visual Words[C]\/\/Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 2021: 15465-15474.","DOI":"10.1109\/CVPR46437.2021.01521"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1155\/2018\/4546896"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"crossref","unstructured":"Wu Q Shen C Liu L What value do explicit high level concepts have in vision to language problems?[C]\/\/Proceedings of the IEEE conference on computer vision and pattern recognition. 2016: 203-212.","DOI":"10.1109\/CVPR.2016.29"},{"key":"e_1_3_2_1_9_1","volume-title":"An integrated deep learning framework for joint segmentation of blood pool and myocardium[J]. Medical image analysis","author":"Du X","year":"2020","unstructured":"Du X, Song Y, Liu Y, An integrated deep learning framework for joint segmentation of blood pool and myocardium[J]. Medical image analysis, 2020, 62: 676-685."},{"key":"e_1_3_2_1_10_1","first-page":"5998","article-title":"Attention is all you need[C]\/\/Proceedings of the Advances in neural information processing systems","volume":"2017","author":"Vaswani A","unstructured":"Vaswani A, Shazeer N, Parmar N, Attention is all you need[C]\/\/Proceedings of the Advances in neural information processing systems. Curran Associates Inc, 2017: 5998-6008.","journal-title":"Curran Associates Inc"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.3390\/app8050739"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.3390\/app9163260"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"crossref","unstructured":"Zhang X. Sun X. Luo Y. RSTNet: Captioning with Adaptive Attention on Visual and Non-Visual Words[C]\/\/Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 2021: 15465-15474.","DOI":"10.1109\/CVPR46437.2021.01521"},{"volume-title":"29th ACM International Conference on Multimedia. 2021: 5056-5064","author":"Song Z","key":"e_1_3_2_1_14_1","unstructured":"Song Z, Zhou X, Dong L, Direction Relation Transformer for Image Captioning[C]\/\/Proceedings of the 29th ACM International Conference on Multimedia. 2021: 5056-5064."},{"volume-title":"Dual graph convolutional networks with transformer and curriculum learning for image captioning[C]\/\/Proceedings of the 29th ACM International Conference on Multimedia. 2021: 2615-2624","author":"Dong X","key":"e_1_3_2_1_15_1","unstructured":"Dong X, Long C, Xu W, Dual graph convolutional networks with transformer and curriculum learning for image captioning[C]\/\/Proceedings of the 29th ACM International Conference on Multimedia. 2021: 2615-2624."},{"issue":"6","key":"e_1_3_2_1_16_1","first-page":"230","article-title":"Transforming objects into words[J]","volume":"32","author":"Herdade S","year":"2019","unstructured":"Herdade S, Kappeler A, Boakye K, Image captioning: Transforming objects into words[J]. Advances in Neural Information Processing Systems, 2019, 32(6): 230-242.","journal-title":"Advances in Neural Information Processing Systems"},{"volume-title":"IEEE\/CVF International Conference on Computer Vision. 2021: 3139-3143","author":"Zhou Y","key":"e_1_3_2_1_17_1","unstructured":"Zhou Y, Zhang Y, Hu Z, Semi-Autoregressive Transformer for Image Captioning[C]\/\/Proceedings of the IEEE\/CVF International Conference on Computer Vision. 2021: 3139-3143."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"crossref","unstructured":"Cornia M Stefanini M Baraldi L Meshed-memory transformer for image captioning[C]\/\/Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 2020: 10578-10587.","DOI":"10.1109\/CVPR42600.2020.01059"},{"volume-title":"Proceedings of the IEEE conference on computer vision and pattern recognition. (2015)","author":"Karpathy A.","key":"e_1_3_2_1_19_1","unstructured":"Karpathy, A., Fei-Fei, L.: Deep visual-semantic alignments for generating imagedescriptions. In: Proceedings of the IEEE conference on computer vision and pattern recognition. (2015) 3128\u20133137"},{"key":"e_1_3_2_1_20_1","first-page":"318","volume-title":"40th Annual Meeting on Association for Computational Linguistics. Association for Computational Linguistics","year":"2002","unstructured":"the 40th Annual Meeting on Association for Computational Linguistics. Association for Computational Linguistics, Philadelphia, PA, USA, 7\u201312 July 2002; pp. 311\u2013318."},{"key":"e_1_3_2_1_21_1","volume-title":"Proceedings of the Workshop on Text Summarization Branches Out","author":"Flick C. ROUGE","year":"2004","unstructured":"Flick, C. ROUGE: A Package for Automatic Evaluation of summaries. In Proceedings of the Workshop on Text Summarization Branches Out, Barcelona, Spain, 25\u201326 July 2004."},{"key":"e_1_3_2_1_22_1","volume-title":"CIDEr: Consensus-based Image Description Evaluation.In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"Vedantam R.","year":"2015","unstructured":"Vedantam, R.; Zitnick, C.L.; Parikh, D. CIDEr: Consensus-based Image Description Evaluation.In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, Boston, MA, USA,7\u201312 June 2015."},{"volume-title":"Proceedings of the Second Workshop on Statistical Machine Translation. Association for Computational Linguistics, Prague, Czech Republic","author":"Lavie A.","key":"e_1_3_2_1_23_1","unstructured":"Lavie, A.; Agarwal, A. METEOR: An automatic metric for MT evaluation with high levels of correlation with human judgments. In Proceedings of the Second Workshop on Statistical Machine Translation. Association for Computational Linguistics, Prague, Czech Republic, 23 June 2007; pp. 228\u2013231."},{"key":"e_1_3_2_1_24_1","volume-title":"Ranger21: a synergistic deep learning optimizer[J]. arXiv preprint arXiv:2106.13731","author":"Wright L","year":"2021","unstructured":"Wright L, Demeure N. Ranger21: a synergistic deep learning optimizer[J]. arXiv preprint arXiv:2106.13731, 2021."},{"volume-title":"3rd International Conference on Natural Language Processing (ICNLP). IEEE","author":"He C","key":"e_1_3_2_1_25_1","unstructured":"He C, Tong Q, Yang X, A Multi-layer Feature Parallel Processing Method for Image Captioning[C]\/\/2021 3rd International Conference on Natural Language Processing (ICNLP). IEEE, 2021: 255-261."}],"event":{"name":"AIPR 2022: 2022 5th International Conference on Artificial Intelligence and Pattern Recognition","acronym":"AIPR 2022","location":"Xiamen China"},"container-title":["Proceedings of the 2022 5th International Conference on Artificial Intelligence and Pattern Recognition"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3573942.3574052","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3573942.3574052","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T19:02:32Z","timestamp":1750186952000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3573942.3574052"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,9,23]]},"references-count":25,"alternative-id":["10.1145\/3573942.3574052","10.1145\/3573942"],"URL":"https:\/\/doi.org\/10.1145\/3573942.3574052","relation":{},"subject":[],"published":{"date-parts":[[2022,9,23]]},"assertion":[{"value":"2023-05-16","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}