{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,22]],"date-time":"2026-01-22T06:34:49Z","timestamp":1769063689015,"version":"3.49.0"},"publisher-location":"New York, NY, USA","reference-count":67,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,8,11]],"date-time":"2023-08-11T00:00:00Z","timestamp":1691712000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,8,11]]},"DOI":"10.1145\/3617695.3617717","type":"proceedings-article","created":{"date-parts":[[2023,11,2]],"date-time":"2023-11-02T22:10:15Z","timestamp":1698963015000},"page":"47-56","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Enhanced feature extraction-based semantic segmentation network for remote sensing image using modified Swin Transformer"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-5798-3740","authenticated-orcid":false,"given":"Song","family":"Peng","sequence":"first","affiliation":[{"name":"Chongqing Technology and Business University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-9402-5000","authenticated-orcid":false,"given":"Gui Liang","family":"Tang","sequence":"additional","affiliation":[{"name":"Chongqing Technology and Business University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-9165-4150","authenticated-orcid":false,"given":"Xue","family":"Wang","sequence":"additional","affiliation":[{"name":"Chongqing Technology and Business University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-7120-4004","authenticated-orcid":false,"given":"Da Hong","family":"Mou","sequence":"additional","affiliation":[{"name":"Chongqing Technology and Business University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-3733-2994","authenticated-orcid":false,"given":"Ling Xiu","family":"Zhu","sequence":"additional","affiliation":[{"name":"Chongqing Technology and Business University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-7919-1320","authenticated-orcid":false,"given":"Xuan","family":"Lai","sequence":"additional","affiliation":[{"name":"Chongqing Technology and Business University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,11,2]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Cao Y","author":"Liu Z","year":"2021","unstructured":"Liu Z , Lin Y , Cao Y , Swin Transformer : Hierarchical Vision Transformer using Shifted Windows [J]. 2021 . Liu Z , Lin Y , Cao Y , Swin Transformer: Hierarchical Vision Transformer using Shifted Windows[J]. 2021."},{"key":"e_1_3_2_1_2_1","volume-title":"LoveDA: A Remote Sensing Land-Cover Dataset for Domain Adaptive Semantic Segmentation[J]","author":"Wang J","year":"2021","unstructured":"Wang J , Zheng Z , Ma A , LoveDA: A Remote Sensing Land-Cover Dataset for Domain Adaptive Semantic Segmentation[J] . 2021 . Wang J , Zheng Z , Ma A , LoveDA: A Remote Sensing Land-Cover Dataset for Domain Adaptive Semantic Segmentation[J]. 2021."},{"issue":"11","key":"e_1_3_2_1_3_1","first-page":"2019","article-title":"An active deep learning approach for minimally supervised POLSAR image classification","volume":"57","author":"Bi H.","unstructured":"H. Bi , F. Xu , Z. Wei , Y . Xue , and Z. Xu , \u201c An active deep learning approach for minimally supervised POLSAR image classification ,\u201d IEEE Trans. Geosci. Remote Sens. , vol. 57 , no. 11 , pp. Nov. 2019 . H. Bi, F. Xu, Z. Wei, Y . Xue, and Z. Xu, \u201cAn active deep learning approach for minimally supervised POLSAR image classification,\u201d IEEE Trans. Geosci. Remote Sens., vol. 57, no. 11, pp. Nov. 2019.","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"e_1_3_2_1_4_1","volume-title":"Knowl.-Based Syst.","author":"Liu X.","unstructured":"X. Liu , L. Jiao , L. Li , X. Tang , and Y . Guo , \u201c Deep multi-level fusion network for multi-source image pixel-wise classification ,\u201d Knowl.-Based Syst. , vol. Jun. Art. no. 106921 . X. Liu, L. Jiao, L. Li, X. Tang, and Y . Guo, \u201cDeep multi-level fusion network for multi-source image pixel-wise classification,\u201d Knowl.-Based Syst., vol. Jun. Art. no. 106921."},{"issue":"12","key":"e_1_3_2_1_5_1","first-page":"2020","article-title":"MS-RRFSegNet: Multiscale regional relation feature segmentation network for semantic segmentation of urban scene point clouds","volume":"58","author":"Luo H.","unstructured":"H. Luo , C. Chen , L. Fang , K. Khoshelham , and G. Shen , \u201c MS-RRFSegNet: Multiscale regional relation feature segmentation network for semantic segmentation of urban scene point clouds ,\u201d IEEE Trans. Geosci. Remote Sens. , vol. 58 , no. 12 , pp. Dec. 2020 . H. Luo, C. Chen, L. Fang, K. Khoshelham, and G. Shen, \u201cMS-RRFSegNet: Multiscale regional relation feature segmentation network for semantic segmentation of urban scene point clouds,\u201d IEEE Trans. Geosci. Remote Sens., vol. 58, no. 12, pp. Dec. 2020.","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"issue":"6","key":"e_1_3_2_1_6_1","first-page":"2021","article-title":"Multistage fusion and multi-source attention network for multi-modal remote sensing image segmentation","volume":"12","author":"Zhao J.","unstructured":"J. Zhao , Y . Zhou , B. Shi , J. Y ang, D. Zhang , and R. Y ao , \u201c Multistage fusion and multi-source attention network for multi-modal remote sensing image segmentation ,\u201d ACM Trans. Intell. Syst. Technol. , vol. 12 , no. 6 , pp. Dec. 2021 . J. Zhao, Y . Zhou, B. Shi, J. Y ang, D. Zhang, and R. Y ao, \u201cMultistage fusion and multi-source attention network for multi-modal remote sensing image segmentation,\u201d ACM Trans. Intell. Syst. Technol., vol. 12, no. 6, pp. Dec. 2021.","journal-title":"ACM Trans. Intell. Syst. Technol."},{"issue":"8","key":"e_1_3_2_1_7_1","first-page":"2020","article-title":"Semantic segmentation of largesize VHR remote sensing images using a two-stage multiscale training architecture","volume":"58","author":"Ding L.","unstructured":"L. Ding , J. Zhang , and L. Bruzzone , \u201c Semantic segmentation of largesize VHR remote sensing images using a two-stage multiscale training architecture ,\u201d IEEE Trans. Geosci. Remote Sens. , vol. 58 , no. 8 , pp. Aug. 2020 . L. Ding, J. Zhang, and L. Bruzzone, \u201cSemantic segmentation of largesize VHR remote sensing images using a two-stage multiscale training architecture,\u201d IEEE Trans. Geosci. Remote Sens., vol. 58, no. 8, pp. Aug. 2020.","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2020.2992177"},{"issue":"9","key":"e_1_3_2_1_9_1","first-page":"2010","article-title":"Using aerial imagery and gis in automated building footprint extraction and shape recognition for earthquake risk assessment of urban inventories","volume":"48","author":"Sahar L.","unstructured":"L. Sahar , S. Muthukumar , and S. P . French , \u201c Using aerial imagery and gis in automated building footprint extraction and shape recognition for earthquake risk assessment of urban inventories ,\u201d IEEE Trans. Geosci.Remote Sens. , vol. 48 , no. 9 , pp. Sep. 2010 . L. Sahar, S. Muthukumar, and S. P . French, \u201cUsing aerial imagery and gis in automated building footprint extraction and shape recognition for earthquake risk assessment of urban inventories,\u201d IEEE Trans. Geosci.Remote Sens., vol. 48, no. 9, pp. Sep. 2010.","journal-title":"IEEE Trans. Geosci.Remote Sens."},{"key":"e_1_3_2_1_10_1","volume-title":"Pattern Recognit.","author":"Liu G.","unstructured":"G. Liu , L. Li , L. Jiao , Y . Dong , and X. Li , \u201c Stacked Fisher autoencoder for SAR change detection ,\u201d Pattern Recognit. , vol. Dec. Art. no. 106971 . G. Liu, L. Li, L. Jiao, Y . Dong, and X. Li, \u201cStacked Fisher autoencoder for SAR change detection,\u201d Pattern Recognit., vol. Dec. Art. no. 106971."},{"key":"e_1_3_2_1_11_1","volume-title":"Remote Sens.","volume":"13","author":"Y","unstructured":"Y . Y u , \u201c Crop row segmentation and detection in paddy fields based on treble-classification Otsu and double-dimensional clustering method ,\u201d Remote Sens. , vol. 13 , no. 5, p. Feb. 2021. Y . Y u , \u201cCrop row segmentation and detection in paddy fields based on treble-classification Otsu and double-dimensional clustering method,\u201d Remote Sens., vol. 13, no. 5, p. Feb. 2021."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA40945.2020.9196722"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298965"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10462-020-09854-1"},{"key":"e_1_3_2_1_15_1","first-page":"05566","article-title":"Image segmentation using deep learning: A survey","volume":"2001","author":"Minaee S.","year":"2020","unstructured":"S. Minaee , Y . Boykov , F. Porikli , A. Plaza , N. Kehtarnavaz , and D. Terzopoulos , \u201c Image segmentation using deep learning: A survey ,\u201d CoRR , vol. abs\/ 2001 . 05566 , pp. Jan. 2020 . S. Minaee, Y . Boykov, F. Porikli, A. Plaza, N. Kehtarnavaz, and D. Terzopoulos, \u201cImage segmentation using deep learning: A survey,\u201d CoRR, vol. abs\/2001.05566, pp. Jan. 2020.","journal-title":"CoRR"},{"key":"e_1_3_2_1_16_1","first-page":"241","volume-title":"Proc. 18th Int. Conf.Med. Image Comput. Comput.-Assist. Intervent.","volume":"9351","author":"Fischer P.","year":"2015","unstructured":"Ronneberger, P. Fischer , and T. Brox , \u201c U-Net: Convolutional networks for biomedical image segmentation ,\u201d in Proc. 18th Int. Conf.Med. Image Comput. Comput.-Assist. Intervent. , vol. 9351 , 2015 , pp. 234\u2013 241 . Ronneberger, P. Fischer, and T. Brox, \u201cU-Net: Convolutional networks for biomedical image segmentation,\u201d in Proc. 18th Int. Conf.Med. Image Comput. Comput.-Assist. Intervent., vol. 9351, 2015, pp. 234\u2013241."},{"key":"e_1_3_2_1_17_1","volume-title":"Encoderdecoder with atrous separable convolution for semantic image segmentation","author":"Chen L.-C.","year":"2018","unstructured":"L.-C. Chen , Y . Zhu , G. Papandreou , F. Schroff , and H. Adam , \u201c Encoderdecoder with atrous separable convolution for semantic image segmentation ,\u201d 2018 , arXiv: 1802.02611. L.-C. Chen, Y . Zhu, G. Papandreou, F. Schroff, and H. Adam, \u201cEncoderdecoder with atrous separable convolution for semantic image segmentation,\u201d 2018, arXiv:1802.02611."},{"key":"e_1_3_2_1_18_1","volume-title":"Rethinking atrous convolution for semantic image segmentation","author":"Chen L.-C.","year":"2017","unstructured":"L.-C. Chen , G. Papandreou , F. Schroff , and H. Adam , \u201c Rethinking atrous convolution for semantic image segmentation ,\u201d 2017 , arXiv: 1706.05587. L.-C. Chen, G. Papandreou, F. Schroff, and H. Adam, \u201cRethinking atrous convolution for semantic image segmentation,\u201d 2017, arXiv:1706.05587."},{"key":"e_1_3_2_1_19_1","first-page":"65356","volume-title":"DenseU-net-based semantic segmentation of small objects in urban remote sensing imagess","author":"Dong R.","year":"2019","unstructured":"R. Dong , X. Pan , and F. Li , \u201c DenseU-net-based semantic segmentation of small objects in urban remote sensing imagess ,\u201d IEEE Access , v o l . 7 , pp. 65347\u2013 65356 , 2019 . R. Dong, X. Pan, and F. Li, \u201cDenseU-net-based semantic segmentation of small objects in urban remote sensing imagess,\u201d IEEE Access, v o l . 7 , pp. 65347\u201365356, 2019."},{"issue":"4","key":"e_1_3_2_1_20_1","first-page":"2021","article-title":"Adaptive effective receptive field convolution for semantic segmentation of VHR remote sensing images","volume":"59","author":"Chen X.","unstructured":"X. Chen , \u201c Adaptive effective receptive field convolution for semantic segmentation of VHR remote sensing images ,\u201d IEEE Trans.Geosci. Remote Sens. , vol. 59 , no. 4 , pp. Apr. 2021 . X. Chen , \u201cAdaptive effective receptive field convolution for semantic segmentation of VHR remote sensing images,\u201d IEEE Trans.Geosci. Remote Sens., vol. 59, no. 4, pp. Apr. 2021.","journal-title":"IEEE Trans.Geosci. Remote Sens."},{"issue":"1","key":"e_1_3_2_1_21_1","first-page":"2021","article-title":"LANet: Local attention embedding to improve the semantic segmentation of remote sensing images","volume":"59","author":"Ding L.","unstructured":"L. Ding , H. Tang , and L. Bruzzone , \u201c LANet: Local attention embedding to improve the semantic segmentation of remote sensing images ,\u201d IEEE Trans. Geosci. Remote Sens. , vol. 59 , no. 1 , pp. Jan. 2021 . L. Ding, H. Tang, and L. Bruzzone, \u201cLANet: Local attention embedding to improve the semantic segmentation of remote sensing images,\u201d IEEE Trans. Geosci. Remote Sens., vol. 59, no. 1, pp. Jan. 2021.","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"e_1_3_2_1_22_1","first-page":"04306","article-title":"TransUNet: Transformers make strong encoders for medical image segmentation","volume":"2102","author":"Chen J.","year":"2021","unstructured":"J. Chen , \u201c TransUNet: Transformers make strong encoders for medical image segmentation ,\u201d CoRR , vol. abs\/ 2102 . 04306 , pp. Feb. 2021 . J. Chen , \u201cTransUNet: Transformers make strong encoders for medical image segmentation,\u201d CoRR, vol. abs\/2102.04306, pp. Feb. 2021.","journal-title":"CoRR"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00326"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00069"},{"key":"e_1_3_2_1_25_1","volume-title":"Pattern Recognit.","volume":"95","author":"Zhao B.","unstructured":"B. Zhao , L. Hua , X. Li , X. Lu , and Z. Wang , \u201c Weather recognition via classification labels and weather-cue maps ,\u201d Pattern Recognit. , vol. 95 , pp. Oct. B. Zhao, L. Hua, X. Li, X. Lu, and Z. Wang, \u201cWeather recognition via classification labels and weather-cue maps,\u201d Pattern Recognit., vol. 95, pp. Oct."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.660"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01228-1_26"},{"issue":"11","key":"e_1_3_2_1_28_1","first-page":"2020","article-title":"Relation matters: Relational contextaware fully convolutional network for semantic segmentation of highresolution aerial images","volume":"58","author":"Mou L.","unstructured":"L. Mou , Y . Hua , and X. X. Zhu , \u201c Relation matters: Relational contextaware fully convolutional network for semantic segmentation of highresolution aerial images ,\u201d IEEE Trans. Geosci. Remote. Sens. , vol. 58 , no. 11 , pp. Dec. 2020 . L. Mou, Y . Hua, and X. X. Zhu, \u201cRelation matters: Relational contextaware fully convolutional network for semantic segmentation of highresolution aerial images,\u201d IEEE Trans. Geosci. Remote. Sens., vol. 58, no. 11, pp. Dec. 2020.","journal-title":"IEEE Trans. Geosci. Remote. Sens."},{"key":"e_1_3_2_1_29_1","first-page":"5","volume-title":"Proc. 9th Int. Conf. Learn. Represent.2021","author":"Dosovitskiy","unstructured":"Dosovitskiy , \u201c An image is worth 16\u00d716 words: Transformers for image recognition at scale ,\u201d in Proc. 9th Int. Conf. Learn. Represent.2021 , pp. 1\u2013 5 . Dosovitskiy , \u201cAn image is worth 16\u00d716 words: Transformers for image recognition at scale,\u201d in Proc. 9th Int. Conf. Learn. Represent.2021, pp. 1\u20135."},{"key":"e_1_3_2_1_30_1","first-page":"229","volume-title":"Proc.ECCV","volume":"2020","author":"Carion N.","unstructured":"N. Carion , F. Massa , G. Synnaeve , N. Usunier , A. Kirillov , and S. Zagoruyko , \u201c End-to-end object detection with transformers ,\u201d in Proc.ECCV , vol. Glasgow, U.K.: Springer, Aug. 2020 , pp. 213\u2013 229 . N. Carion, F. Massa, G. Synnaeve, N. Usunier, A. Kirillov, and S. Zagoruyko, \u201cEnd-to-end object detection with transformers,\u201d in Proc.ECCV, vol. Glasgow, U.K.: Springer, Aug. 2020, pp. 213\u2013229."},{"key":"e_1_3_2_1_31_1","first-page":"14899","article-title":"CrossViT: Cross-attention multi-scale vision transformer for image classification","volume":"2103","author":"Chen C.","year":"2021","unstructured":"C. Chen , Q. Fan , and R. Panda , \u201c CrossViT: Cross-attention multi-scale vision transformer for image classification ,\u201d CoRR , vol. abs\/ 2103 . 14899 , pp. Mar. 2021 . C. Chen, Q. Fan, and R. Panda, \u201cCrossViT: Cross-attention multi-scale vision transformer for image classification,\u201d CoRR, vol. abs\/2103.14899, pp. Mar. 2021.","journal-title":"CoRR"},{"key":"e_1_3_2_1_32_1","first-page":"14030","article-title":"Swin transformer: Hierarchical vision transformer using shifted windows","volume":"2103","author":"Liu Z.","year":"2021","unstructured":"Z. Liu , \u201c Swin transformer: Hierarchical vision transformer using shifted windows ,\u201d CoRR , vol. abs\/ 2103 . 14030 , pp. Mar. 2021 . Z. Liu , \u201cSwin transformer: Hierarchical vision transformer using shifted windows,\u201d CoRR, vol. abs\/2103.14030, pp. Mar. 2021.","journal-title":"CoRR"},{"key":"e_1_3_2_1_33_1","first-page":"05537","article-title":"Swin-Unet: Unet-like pure transformer for medical image segmentation","volume":"2105","author":"Cao H.","year":"2021","unstructured":"H. Cao , \u201c Swin-Unet: Unet-like pure transformer for medical image segmentation ,\u201d CoRR , vol. abs\/ 2105 . 05537 , pp. 1\u201314, May 2021 . H. Cao , \u201cSwin-Unet: Unet-like pure transformer for medical image segmentation,\u201d CoRR, vol. abs\/2105.05537, pp. 1\u201314, May 2021.","journal-title":"CoRR"},{"key":"e_1_3_2_1_34_1","volume-title":"CoRR","author":"Chen B.","unstructured":"Lin, B. Chen , J. Xu , Z. Zhang , and G. Lu , \u201c DS-TransUNet: Dual swin transformer U-Net for medical image segmentation ,\u201d CoRR , vol. abs\/ 2106 .06716, pp. Jun. Lin, B. Chen, J. Xu, Z. Zhang, and G. Lu, \u201cDS-TransUNet: Dual swin transformer U-Net for medical image segmentation,\u201d CoRR, vol. abs\/2106.06716, pp. Jun."},{"key":"e_1_3_2_1_35_1","volume-title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale[J]","author":"Dosovitskiy A","year":"2020","unstructured":"Dosovitskiy A , Beyer L , Kolesnikov A , An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale[J] . 2020 . Dosovitskiy A , Beyer L , Kolesnikov A , An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale[J]. 2020."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2012.231"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2016.2572683"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2005.852470"},{"key":"e_1_3_2_1_39_1","volume-title":"ICML","author":"Pinheiro Pedro","year":"2014","unstructured":"Pedro Pinheiro and Ronan Collobert . Recurrent Convolutional Neural Networks for scene labeling . In ICML , 2014 .2 Pedro Pinheiro and Ronan Collobert. Recurrent Convolutional Neural Networks for scene labeling. In ICML, 2014.2"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.518"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2016.2644615"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.549"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.353"},{"key":"e_1_3_2_1_44_1","volume-title":"MICCAI","author":"Fischer P","year":"2015","unstructured":"Ronneberger, P . Fischer , and T. Brox . U-Net: Convolutional networks for biomedical image segmentation . In MICCAI , 2015 . 2 Ronneberger, P .Fischer, and T. Brox. U-Net: Convolutional networks for biomedical image segmentation. In MICCAI, 2015. 2"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2017.2699184"},{"key":"e_1_3_2_1_46_1","volume-title":"ICLR","author":"Liang-Chieh Chen","year":"2015","unstructured":"Chen Liang-Chieh , George Papandreou , Iasonas Kokkinos , kevin murphy, and Alan Y uille. Semantic image segmentation with deep convolutional nets and fully connected CRFs .In ICLR , 2015 . 2, 5 Chen Liang-Chieh, George Papandreou, Iasonas Kokkinos, kevin murphy, and Alan Y uille. Semantic image segmentation with deep convolutional nets and fully connected CRFs.In ICLR, 2015. 2, 5"},{"key":"e_1_3_2_1_47_1","volume-title":"ICLR","author":"Vladlen Koltun Fisher Y","year":"2016","unstructured":"Fisher Y u and Vladlen Koltun . Multi-scale context aggregation by dilated convolutions . In ICLR , 2016 . 2 Fisher Y u and Vladlen Koltun. Multi-scale context aggregation by dilated convolutions. In ICLR, 2016. 2"},{"key":"e_1_3_2_1_48_1","volume-title":"Rethinking atrous convolution for semantic image segmentation. arXiv preprint","author":"Chen Liang-Chieh","year":"2017","unstructured":"Liang-Chieh Chen , George Papandreou , Florian Schroff , and Hartwig Adam . Rethinking atrous convolution for semantic image segmentation. arXiv preprint , 2017 . 2 Liang-Chieh Chen, George Papandreou, Florian Schroff, and Hartwig Adam. Rethinking atrous convolution for semantic image segmentation. arXiv preprint, 2017. 2"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01234-2_49"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.660"},{"key":"e_1_3_2_1_51_1","volume-title":"Advances in Neural Information Processing Systems","volume":"30","author":"Shazeer N.","year":"2017","unstructured":"Vaswani, N. Shazeer , N. Parmar , J. Uszkoreit , L. Jones , A. N. Gomez , L. u. Kaiser , and I. Polosukhin , \u201c Attention is all you need ,\u201d in Advances in Neural Information Processing Systems , vol. 30 . Curran Associates, Inc. , 2017 . Vaswani, N. Shazeer, N. Parmar, J. Uszkoreit, L. Jones, A. N. Gomez, L. u.Kaiser, and I. Polosukhin, \u201cAttention is all you need,\u201d in Advances in Neural Information Processing Systems, vol. 30. Curran Associates, Inc., 2017."},{"key":"e_1_3_2_1_52_1","volume-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","volume":"1","author":"Devlin J.","unstructured":"J. Devlin , M.-W. Chang , K. Lee , and K. Toutanova , \u201c BERT: Pre-training of deep bidirectional transformers for language understanding ,\u201d in Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies , Volume 1 (Long and Short Papers).Minneapolis, Minnesota : Association for Computational Linguistics, Jun. 2019, pp. 4171\u20134186. [Online]. Available: https:\/\/www.aclweb.org\/anthology\/N19-1423 J. Devlin, M.-W. Chang, K. Lee, and K. Toutanova, \u201cBERT: Pre-training of deep bidirectional transformers for language understanding,\u201d in Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers).Minneapolis, Minnesota: Association for Computational Linguistics, Jun. 2019, pp. 4171\u20134186. [Online]. Available: https:\/\/www.aclweb.org\/anthology\/N19-1423"},{"key":"e_1_3_2_1_53_1","first-page":"12877","article-title":"Training data-efficient image transformers & distillation through attention","volume":"2012","author":"Touvron H.","year":"2020","unstructured":". H. Touvron , M. Cord , M. Douze , F. Massa , A. Sablayrolles , and H. J\u00b4 egou , \u201c Training data-efficient image transformers & distillation through attention ,\u201d CoRR , vol.abs\/ 2012 . 12877 , 2020 . . H. Touvron, M. Cord, M. Douze, F. Massa, A. Sablayrolles, and H. J\u00b4 egou, \u201cTraining data-efficient image transformers & distillation through attention,\u201d CoRR, vol.abs\/2012.12877, 2020.","journal-title":"CoRR"},{"key":"e_1_3_2_1_54_1","first-page":"12122","article-title":"Pyramid vision transformer: A versatile backbone for dense prediction without convolutions","volume":"2102","author":"Wang W.","year":"2021","unstructured":"W. Wang , E. Xie , X. Li , D. Fan , K. Song , D. Liang , T. Lu , P. Luo , and L. Shao , \u201c Pyramid vision transformer: A versatile backbone for dense prediction without convolutions ,\u201d CoRR , vol. abs\/ 2102 . 12122 , 2021 . [Online]. Available: https:\/\/arxiv.org\/abs\/2102.12122 W. Wang, E. Xie, X. Li, D. Fan, K. Song, D. Liang, T. Lu, P. Luo, and L. Shao, \u201cPyramid vision transformer: A versatile backbone for dense prediction without convolutions,\u201d CoRR, vol. abs\/2102.12122, 2021. [Online]. Available: https:\/\/arxiv.org\/abs\/2102.12122","journal-title":"CoRR"},{"key":"e_1_3_2_1_55_1","first-page":"00112","article-title":"Transformer in transformer","volume":"2103","author":"Han K.","year":"2021","unstructured":"K. Han , A. Xiao , E. Wu , J. Guo , C. Xu , and Y. Wang , \u201c Transformer in transformer ,\u201d CoRR , vol. abs\/ 2103 . 00112 , 2021 . [Online]. Available: https: \/\/arxiv.org\/abs\/2103.00112 K. Han, A. Xiao, E. Wu, J. Guo, C. Xu, and Y. Wang, \u201cTransformer in transformer,\u201d CoRR, vol. abs\/2103.00112, 2021. [Online]. Available: https: \/\/arxiv.org\/abs\/2103.00112","journal-title":"CoRR"},{"key":"e_1_3_2_1_56_1","first-page":"14030","article-title":"Swin transformer: Hierarchical vision transformer using shifted windows","volume":"2103","author":"Liu Z.","year":"2021","unstructured":"Z. Liu , Y. Lin , Y. Cao , H. Hu , Y. Wei , Z. Zhang , S. Lin , and B. Guo , \u201c Swin transformer: Hierarchical vision transformer using shifted windows ,\u201d CoRR , vol.abs\/ 2103 . 14030 , 2021 Z. Liu, Y. Lin, Y. Cao, H. Hu, Y. Wei, Z. Zhang, S. Lin, and B. Guo, \u201cSwin transformer: Hierarchical vision transformer using shifted windows,\u201d CoRR, vol.abs\/2103.14030, 2021","journal-title":"CoRR"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.3390\/rs9050500"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.rse.2019.111322"},{"key":"e_1_3_2_1_59_1","first-page":"1508","volume-title":"Proc. Winter Conf. Appl. Comput. Vis.","author":"Nigam I.","year":"2018","unstructured":"I. Nigam , C. Huang , and D. Ramanan , \u201c Ensemble knowledge transferfor semantic segmentation ,\u201d in Proc. Winter Conf. Appl. Comput. Vis. , Lake Tahoe, NV, USA , Mar. 2018 , pp. 1499\u2013 1508 . I. Nigam, C. Huang, and D. Ramanan, \u201cEnsemble knowledge transferfor semantic segmentation,\u201d in Proc. Winter Conf. Appl. Comput. Vis.,Lake Tahoe, NV, USA, Mar. 2018, pp. 1499\u20131508."},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.3390\/rs12040701"},{"key":"e_1_3_2_1_61_1","first-page":"04514","article-title":"High-resolution representations for labeling pixels and regions","volume":"1904","author":"Sun K.","year":"2019","unstructured":"K. Sun , \u201c High-resolution representations for labeling pixels and regions ,\u201d CoRR , vol. abs\/ 1904 . 04514 , pp. 1\u201314, Oct. 2019 . K. Sun , \u201cHigh-resolution representations for labeling pixels and regions,\u201d CoRR, vol. abs\/1904.04514, pp. 1\u201314, Oct. 2019.","journal-title":"CoRR"},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2017.2740362"},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/IGARSS.2019.8900507"},{"key":"e_1_3_2_1_64_1","first-page":"688","volume-title":"Proc. Conf. Comput. Vis. Pattern Recognit. Workshops","author":"Kampffmeyer M.","year":"2016","unstructured":"M. Kampffmeyer , A. Salberg , and R. Jenssen , \u201c Semantic segmentationof small objects and modeling of uncertainty in urban remote sensingimages using deep convolutional neural networks ,\u201d in Proc. Conf. Comput. Vis. Pattern Recognit. Workshops , Las Vegas, NV, USA , Jun. 2016 ,pp. 680\u2013 688 . M. Kampffmeyer, A. Salberg, and R. Jenssen, \u201cSemantic segmentationof small objects and modeling of uncertainty in urban remote sensingimages using deep convolutional neural networks,\u201d in Proc. Conf. Comput. Vis. Pattern Recognit. Workshops, Las Vegas, NV, USA, Jun. 2016,pp. 680\u2013688."},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2019.2917952"},{"key":"e_1_3_2_1_66_1","first-page":"5606216","article-title":"FactSeg: Foregroundactivation-driven small object semantic segmentation in large-scaleremote sensing imagery","volume":"60","author":"Ma A.","year":"2022","unstructured":"A. Ma , J. Wang , Y. Zhong , and Z. Zheng , \u201c FactSeg: Foregroundactivation-driven small object semantic segmentation in large-scaleremote sensing imagery ,\u201d IEEE Trans. Geosci. Remote Sens. , vol. 60 , 2022 , Art. no. 5606216 . A. Ma, J. Wang, Y. Zhong, and Z. Zheng, \u201cFactSeg: Foregroundactivation-driven small object semantic segmentation in large-scaleremote sensing imagery,\u201d IEEE Trans. Geosci. Remote Sens., vol. 60,2022, Art. no. 5606216.","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2020.3009143"}],"event":{"name":"BDIOT 2023: 2023 7th International Conference on Big Data and Internet of Things","location":"Beijing China","acronym":"BDIOT 2023"},"container-title":["Proceedings of the 2023 7th International Conference on Big Data and Internet of Things"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3617695.3617717","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3617695.3617717","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T16:37:56Z","timestamp":1750178276000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3617695.3617717"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,8,11]]},"references-count":67,"alternative-id":["10.1145\/3617695.3617717","10.1145\/3617695"],"URL":"https:\/\/doi.org\/10.1145\/3617695.3617717","relation":{},"subject":[],"published":{"date-parts":[[2023,8,11]]},"assertion":[{"value":"2023-11-02","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}