{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,7,8]],"date-time":"2025-07-08T06:40:08Z","timestamp":1751956808863,"version":"3.41.2"},"publisher-location":"New York, NY, USA","reference-count":30,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,11,22]]},"DOI":"10.1145\/3725949.3725958","type":"proceedings-article","created":{"date-parts":[[2025,7,8]],"date-time":"2025-07-08T05:53:12Z","timestamp":1751953992000},"page":"50-55","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Transformer Based Remote Sensing Image Captioning with Object Count Information"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-6056-6477","authenticated-orcid":false,"given":"Zihao","family":"Ni","sequence":"first","affiliation":[{"name":"China University of Petroleum (East China), Qingdao, Shandong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6175-8451","authenticated-orcid":false,"given":"Zhaoyun","family":"Zong","sequence":"additional","affiliation":[{"name":"China University of Petroleum (East China), Qingdao, Shandong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-2919-7142","authenticated-orcid":false,"given":"Yinghao","family":"Xu","sequence":"additional","affiliation":[{"name":"China University of Petroleum (East China), Qingdao, Shandong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3949-985X","authenticated-orcid":false,"given":"Peng","family":"Ren","sequence":"additional","affiliation":[{"name":"China University of Petroleum (East China), Qingdao, Shandong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,7,7]]},"reference":[{"doi-asserted-by":"crossref","unstructured":"Abdullah\u00a0M Algamdi and Hammam\u00a0M Alghamdi. 2023. Instant Counting & Vehicle Detection during Hajj Using Drones. Journal of Image and Graphics 11 2 (2023) 204\u2013211.","key":"e_1_3_3_1_2_2","DOI":"10.18178\/joig.11.2.204-211"},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_3_2","DOI":"10.1007\/978-3-319-46454-124.35"},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_4_2","DOI":"10.3115\/v1\/W14-3348"},{"doi-asserted-by":"publisher","unstructured":"Pejman Gholami-Dastgerdi and Mohammad-Reza Feizi-Derakhshi. 2023. Part of speech tagging using part of speech sequence graph. Annals of Data Science 10 5 (2023) 1301\u20131328. 10.1007\/s40745-021-00359-4","key":"e_1_3_3_1_5_2","DOI":"10.1007\/s40745-021-00359-4"},{"doi-asserted-by":"publisher","unstructured":"Kai Han Yunhe Wang Hanting Chen Xinghao Chen Jianyuan Guo Zhenhua Liu Yehui Tang An Xiao Chunjing Xu Yixing Xu et\u00a0al. 2022. A survey on vision transformer. IEEE transactions on pattern analysis and machine intelligence 45 1 (2022) 87\u2013110. 10.1109\/TMM.2024.3369863","key":"e_1_3_3_1_6_2","DOI":"10.1109\/TMM.2024.3369863"},{"key":"e_1_3_3_1_7_2","first-page":"74","volume-title":"Text summarization branches out","author":"Lin Chin-Yew","year":"2004","unstructured":"Chin-Yew Lin. 2004. Rouge: A package for automatic evaluation of summaries. In Text summarization branches out. 74\u201381."},{"doi-asserted-by":"publisher","unstructured":"Chenyang Liu Rui Zhao Hao Chen Zhengxia Zou and Zhenwei Shi. 2022. Remote sensing image change captioning with dual-branch transformers: A new method and a large scale dataset. IEEE Transactions on Geoscience and Remote Sensing 60 (2022) 1\u201320. 10.1109\/TGRS.2022.3218921","key":"e_1_3_3_1_8_2","DOI":"10.1109\/TGRS.2022.3218921"},{"doi-asserted-by":"publisher","unstructured":"Chenyang Liu Rui Zhao and Zhenwei Shi. 2022. Remote-sensing image captioning based on multilayer aggregated transformer. IEEE Geoscience and Remote Sensing Letters 19 (2022) 1\u20135. 10.1109\/LGRS.2022.3150957","key":"e_1_3_3_1_9_2","DOI":"10.1109\/LGRS.2022.3150957"},{"doi-asserted-by":"publisher","unstructured":"Yue Ming Nannan Hu Chunxiao Fan Fan Feng Jiangwan Zhou and Hui Yu. 2022. Visuals to text: A comprehensive review on automatic image captioning. IEEE\/CAA Journal of Automatica Sinica 9 8 (2022) 1339\u20131365. 10.1109\/JAS.2022.105734","key":"e_1_3_3_1_10_2","DOI":"10.1109\/JAS.2022.105734"},{"doi-asserted-by":"publisher","unstructured":"Zihao Ni Zhaoyun Zong and Peng Ren. 2024. Incorporating object counts into remote sensing image captioning. International Journal of Digital Earth 17 1 (2024) 2392847. 10.1080\/17538947.2024.2392847","key":"e_1_3_3_1_11_2","DOI":"10.1080\/17538947.2024.2392847"},{"key":"e_1_3_3_1_12_2","first-page":"311","volume-title":"Proceedings of the 40th annual meeting of the Association for Computational Linguistics","author":"Papineni Kishore","year":"2002","unstructured":"Kishore Papineni, Salim Roukos, Todd Ward, and Wei-Jing Zhu. 2002. Bleu: a method for automatic evaluation of machine translation. In Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311\u2013318."},{"doi-asserted-by":"publisher","unstructured":"Umme\u00a0Fawzia Rahim and Hiroshi Mineno. 2020. Tomato flower detection and counting in greenhouses using faster region-based convolutional neural network. Journal of Image and Graphics 8 4 (2020) 107\u2013113. 10.18178\/joig.8.4.107-113","key":"e_1_3_3_1_13_2","DOI":"10.18178\/joig.8.4.107-113"},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_14_2","DOI":"10.1109\/CVPR52729.2023.00278"},{"doi-asserted-by":"crossref","unstructured":"Abdul\u00a0Haris Rangkuti Varyl\u00a0Hasbi Athala and Farrel\u00a0Haridhi Indallah. 2023. Development of vehicle detection and counting systems with uav cameras: Deep learning and darknet algorithms. Journal of Image and Graphics 11 3 (2023).","key":"e_1_3_3_1_15_2","DOI":"10.18178\/joig.11.3.248-262"},{"doi-asserted-by":"publisher","unstructured":"Zihao Ren Shuiping Gou Zhang Guo Shasha Mao and Ruimin Li. 2022. A mask-guided transformer network with topic token for remote sensing image captioning. Remote Sensing 14 12 (2022) 2939. 10.3390\/rs14122939","key":"e_1_3_3_1_16_2","DOI":"10.3390\/rs14122939"},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_17_2","DOI":"10.1109\/CVPR.2017.131"},{"doi-asserted-by":"crossref","unstructured":"Zhuang Shao Jungong Han Kurt Debattista and Yanwei Pang. 2024. DCMSTRD: end-to-end dense captioning via multi-scale transformer decoding. IEEE Transactions on Multimedia (2024).","key":"e_1_3_3_1_18_2","DOI":"10.1109\/TMM.2024.3369863"},{"doi-asserted-by":"publisher","unstructured":"Himanshu Sharma and Devanand Padha. 2023. A comprehensive survey on image captioning: from handcrafted to deep learning-based techniques a taxonomy and open research issues. Artificial Intelligence Review 56 11 (2023) 13619\u201313661. 10.1007\/s10462-023-10488-2","key":"e_1_3_3_1_19_2","DOI":"10.1007\/s10462-023-10488-2"},{"doi-asserted-by":"publisher","unstructured":"Qian Shi Da He Zhengyu Liu Xiaoping Liu and Jingqian Xue. 2023. Globe230k: A benchmark dense-pixel annotation dataset for global land cover mapping. Journal of Remote Sensing 3 (2023) 0078. 10.34133\/remotesensing.0078","key":"e_1_3_3_1_20_2","DOI":"10.34133\/remotesensing.0078"},{"doi-asserted-by":"publisher","unstructured":"Jia\u00a0Huei Tan Chee\u00a0Seng Chan and Joon\u00a0Huang Chuah. 2022. End-to-end supermask pruning: Learning to prune image captioning models. Pattern Recognition 122 (2022) 108366. 10.1016\/j.patcog.2021.108366","key":"e_1_3_3_1_21_2","DOI":"10.1016\/j.patcog.2021.108366"},{"doi-asserted-by":"publisher","unstructured":"Weixuan Tang Bin Li Weixiang Li Yuangen Wang and Jiwu Huang. 2023. Reinforcement learning of non-additive joint steganographic embedding costs with attention mechanism. Science China Information Sciences 66 3 (2023) 132305. 10.1007\/s11432-021-3453-5","key":"e_1_3_3_1_22_2","DOI":"10.1007\/s11432-021-3453-5"},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_23_2","DOI":"10.1109\/CVPR.2015.7299087"},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_24_2","DOI":"10.1109\/CVPR.2015.7298935"},{"doi-asserted-by":"publisher","unstructured":"Shuang Wang Xiutiao Ye Yu Gu Jihui Wang Yun Meng Jingxian Tian Biao Hou and Licheng Jiao. 2022. Multi-label semantic feature fusion for remote sensing image captioning. ISPRS Journal of Photogrammetry and Remote Sensing 184 (2022) 1\u201318. 10.1016\/j.isprsjprs.2021.11.020","key":"e_1_3_3_1_25_2","DOI":"10.1016\/j.isprsjprs.2021.11.020"},{"doi-asserted-by":"publisher","unstructured":"Yong Wang Wenkai Zhang Zhengyuan Zhang Xin Gao and Xian Sun. 2022. Multiscale multiinteraction network for remote sensing image captioning. IEEE Journal of Selected Topics in Applied Earth Observations and Remote Sensing 15 (2022) 2154\u20132165. 10.1109\/JSTARS.2022.3153636","key":"e_1_3_3_1_26_2","DOI":"10.1109\/JSTARS.2022.3153636"},{"doi-asserted-by":"publisher","unstructured":"Qiaoqiao Yang Zihao Ni and Peng Ren. 2022. Meta captioning: A meta learning based remote sensing image captioning framework. ISPRS Journal of Photogrammetry and Remote Sensing 186 (2022) 190\u2013200. 10.1016\/j.isprsjprs.2022.02.001.115","key":"e_1_3_3_1_27_2","DOI":"10.1016\/j.isprsjprs.2022.02.001.115"},{"doi-asserted-by":"publisher","unstructured":"Xiangrong Zhang Yunpeng Li Xin Wang Feixiang Liu Zhaoji Wu Xina Cheng and Licheng Jiao. 2023. Multi-source interactive stair attention for remote sensing image captioning. Remote Sensing 15 3 (2023) 579. 10.3390\/rs15030579","key":"e_1_3_3_1_28_2","DOI":"10.3390\/rs15030579"},{"doi-asserted-by":"publisher","unstructured":"Rui Zhao Zhenwei Shi and Zhengxia Zou. 2021. High-resolution remote sensing image captioning based on structured attention. IEEE Transactions on Geoscience and Remote Sensing 60 (2021) 1\u201314. 10.1109\/TGRS.2021.3070383","key":"e_1_3_3_1_29_2","DOI":"10.1109\/TGRS.2021.3070383"},{"doi-asserted-by":"publisher","unstructured":"Haonan Zhou Xiaoping Du Lurui Xia and Sen Li. 2022. Self-learning for few-shot remote sensing image captioning. Remote Sensing 14 18 (2022) 4606. 10.3390\/rs14184606","key":"e_1_3_3_1_30_2","DOI":"10.3390\/rs14184606"},{"doi-asserted-by":"publisher","unstructured":"Usman Zia M\u00a0Mohsin Riaz and Abdul Ghafoor. 2022. Transforming remote sensing images to textual descriptions. International Journal of Applied Earth Observation and Geoinformation 108 (2022) 102741. 10.1016\/j.jag.2022.102741","key":"e_1_3_3_1_31_2","DOI":"10.1016\/j.jag.2022.102741"}],"event":{"acronym":"SSIP 2024","name":"SSIP 2024: 2024 7th International Conference on Sensors, Signal and Image Processing","location":"Shenzhen China"},"container-title":["Proceedings of the 2024 7th International Conference on Sensors, Signal and Image Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3725949.3725958","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,8]],"date-time":"2025-07-08T06:20:31Z","timestamp":1751955631000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3725949.3725958"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,22]]},"references-count":30,"alternative-id":["10.1145\/3725949.3725958","10.1145\/3725949"],"URL":"https:\/\/doi.org\/10.1145\/3725949.3725958","relation":{},"subject":[],"published":{"date-parts":[[2024,11,22]]},"assertion":[{"value":"2025-07-07","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}