{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,19]],"date-time":"2026-03-19T16:35:19Z","timestamp":1773938119877,"version":"3.50.1"},"reference-count":47,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2025,2,15]],"date-time":"2025-02-15T00:00:00Z","timestamp":1739577600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,2,15]],"date-time":"2025-02-15T00:00:00Z","timestamp":1739577600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"IOT Technology Application Transportation Industry R & D Center","award":["202304"],"award-info":[{"award-number":["202304"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2025,5]]},"DOI":"10.1007\/s10489-025-06337-0","type":"journal-article","created":{"date-parts":[[2025,2,15]],"date-time":"2025-02-15T04:23:01Z","timestamp":1739593381000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["GANet: geometry-aware network for RGB-D semantic segmentation"],"prefix":"10.1007","volume":"55","author":[{"given":"Chunqi","family":"Tian","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-0102-5021","authenticated-orcid":false,"given":"Weirong","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lizhi","family":"Bai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jun","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yanjun","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,2,15]]},"reference":[{"issue":"3","key":"6337_CR1","doi-asserted-by":"publisher","first-page":"1000","DOI":"10.1109\/TASE.2020.2993143","volume":"18","author":"Y Sun","year":"2020","unstructured":"Sun Y, Zuo W, Yun P, Wang H, Liu M (2020) Fuseseg: Semantic segmentation of urban scenes based on rgb and thermal data fusion. IEEE Trans Automat Sci Eng 18(3):1000\u20131011","journal-title":"IEEE Trans Automat Sci Eng"},{"issue":"4","key":"6337_CR2","doi-asserted-by":"publisher","first-page":"1596","DOI":"10.1109\/TASE.2019.2893414","volume":"16","author":"Y Sun","year":"2019","unstructured":"Sun Y, Liu M, Meng MQ-H (2019) Active perception for foreground segmentation: An rgb-d data-based background modeling method. IEEE Trans Automat Sci Eng 16(4):1596\u20131609","journal-title":"IEEE Trans Automat Sci Eng"},{"key":"6337_CR3","doi-asserted-by":"crossref","unstructured":"Zhao H, Shi J, Qi X, Wang X, Jia J (2017) Pyramid scene parsing network. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 2881\u20132890","DOI":"10.1109\/CVPR.2017.660"},{"issue":"12","key":"6337_CR4","doi-asserted-by":"publisher","first-page":"2481","DOI":"10.1109\/TPAMI.2016.2644615","volume":"39","author":"V Badrinarayanan","year":"2017","unstructured":"Badrinarayanan V, Kendall A, Cipolla R (2017) Segnet: A deep convolutional encoder-decoder architecture for image segmentation. IEEE Trans Pattern Anal Mach Intell 39(12):2481\u20132495","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"6337_CR5","unstructured":"Chen L-C, Papandreou G, Kokkinos I, Murphy K, Yuille AL (2014) Semantic image segmentation with deep convolutional nets and fully connected crfs. arXiv:1412.7062"},{"key":"6337_CR6","unstructured":"Park S-J, Hong K-S, Lee S (2017) Rdfnet: Rgb-d multi-level residual feature fusion for indoor semantic segmentation. In: Proceedings of the IEEE International Conference on Computer Vision, pp 4980\u20134989"},{"key":"6337_CR7","doi-asserted-by":"crossref","unstructured":"Giannone G, Chidlovskii B (2019) Learning common representation from rgb and depth images. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops, pp 0\u20130","DOI":"10.1109\/CVPRW.2019.00054"},{"key":"6337_CR8","doi-asserted-by":"crossref","unstructured":"Hu X, Yang K, Fei L, Wang K (2019) Acnet: Attention based network to exploit complementary features for rgbd semantic segmentation. In: 2019 IEEE International Conference on Image Processing (ICIP), pp 1440\u20131444 . IEEE","DOI":"10.1109\/ICIP.2019.8803025"},{"key":"6337_CR9","unstructured":"Bai L, Yang J, Tian C, Sun Y, Mao M, Xu Y, Xu W (2022) Dcanet: Differential convolution attention network for rgb-d semantic segmentation. arXiv:2210.06747"},{"key":"6337_CR10","unstructured":"Yang J, Bai L, Sun Y, Tian C, Mao M, Wang G (2023) Pixel difference convolutional network for rgb-d semantic segmentation. IEEE Trans Circ Syst Video Technol, 1\u20131"},{"key":"6337_CR11","doi-asserted-by":"crossref","unstructured":"Cao J, Leng H, Lischinski D, Cohen-Or D, Tu C, Li Y (2021) Shapeconv: Shape-aware convolutional layer for indoor rgb-d semantic segmentation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 7088\u20137097","DOI":"10.1109\/ICCV48922.2021.00700"},{"key":"6337_CR12","doi-asserted-by":"crossref","unstructured":"Zhang J, Liu H, Yang K, Hu X, Liu R, Stiefelhagen R (2022) Cmx: Cross-modal fusion for rgb-x semantic segmentation with transformers. arXiv:2203.04838","DOI":"10.1109\/TITS.2023.3300537"},{"key":"6337_CR13","doi-asserted-by":"crossref","unstructured":"Silberman N, Hoiem D, Kohli P, Fergus R (2012) Indoor segmentation and support inference from rgbd images. In: Computer Vision\u2013ECCV 2012: 12th European Conference on Computer Vision, Florence, Italy, 7-13 October, 2012, Proceedings, Part V 12, pp 746\u2013760 . Springer","DOI":"10.1007\/978-3-642-33715-4_54"},{"key":"6337_CR14","doi-asserted-by":"crossref","unstructured":"Wang W, Neumann U (2018) Depth-aware cnn for rgb-d segmentation. In: Proceedings of the European Conference on Computer Vision (ECCV), pp 135\u2013150","DOI":"10.1007\/978-3-030-01252-6_9"},{"key":"6337_CR15","doi-asserted-by":"crossref","unstructured":"Chen X, Lin K-Y, Qian C, Zeng G, Li H (2020) 3d sketch-aware semantic scene completion via semi-supervised structure prior. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 4193\u20134202","DOI":"10.1109\/CVPR42600.2020.00425"},{"key":"6337_CR16","doi-asserted-by":"crossref","unstructured":"Wang J, Wang Z, Tao D, See S, Wang G (2016) Learning common and specific features for rgb-d semantic segmentation with deconvolutional networks. In: Computer Vision\u2013ECCV 2016: 14th European Conference, Amsterdam, The Netherlands, 11-14 October, 2016, Proceedings, Part V 14, pp 664\u2013679. Springer","DOI":"10.1007\/978-3-319-46454-1_40"},{"issue":"4\u20135","key":"6337_CR17","doi-asserted-by":"publisher","first-page":"437","DOI":"10.1177\/0278364917713117","volume":"37","author":"M Schwarz","year":"2018","unstructured":"Schwarz M, Milan A, Periyasamy AS, Behnke S (2018) Rgb-d object detection and semantic segmentation for autonomous manipulation in clutter. The Int J Robot Res 37(4\u20135):437\u2013451","journal-title":"The Int J Robot Res"},{"key":"6337_CR18","doi-asserted-by":"publisher","first-page":"2313","DOI":"10.1109\/TIP.2021.3049332","volume":"30","author":"L-Z Chen","year":"2021","unstructured":"Chen L-Z, Lin Z, Wang Z, Yang Y-L, Cheng M-M (2021) Spatial information guided convolution for real-time rgbd semantic segmentation. IEEE Trans Image Process 30:2313\u20132324","journal-title":"IEEE Trans Image Process"},{"issue":"8","key":"6337_CR19","doi-asserted-by":"publisher","first-page":"8735","DOI":"10.1007\/s10489-022-03969-4","volume":"53","author":"Y Lu","year":"2023","unstructured":"Lu Y, Yu H, Ni W, Song L (2023) 3d real-time human reconstruction with a single rgbd camera. Appl Intell 53(8):8735\u20138745","journal-title":"Appl Intell"},{"issue":"12","key":"6337_CR20","doi-asserted-by":"publisher","first-page":"14838","DOI":"10.1007\/s10489-022-04179-8","volume":"53","author":"L Yu","year":"2023","unstructured":"Yu L, Tian L, Du Q, Bhutto JA (2023) Multi-stream adaptive 3d attention graph convolution network for skeleton-based action recognition. Appl Intell 53(12):14838\u201314854","journal-title":"Appl Intell"},{"issue":"12","key":"6337_CR21","doi-asserted-by":"publisher","first-page":"15390","DOI":"10.1007\/s10489-022-04302-9","volume":"53","author":"S Chen","year":"2023","unstructured":"Chen S, Xu K, Zhu B, Jiang X, Sun T (2023) Deformable graph convolutional transformer for skeleton-based action recognition. Appl Intell 53(12):15390\u201315406","journal-title":"Appl Intell"},{"issue":"15","key":"6337_CR22","doi-asserted-by":"publisher","first-page":"18167","DOI":"10.1007\/s10489-022-03401-x","volume":"52","author":"T Gao","year":"2022","unstructured":"Gao T, Wei W, Cai Z, Fan Z, Xie SQ, Wang X, Yu Q (2022) Ci-net: A joint depth estimation and semantic segmentation network using contextual information. Appl Intell 52(15):18167\u201318186","journal-title":"Appl Intell"},{"key":"6337_CR23","doi-asserted-by":"crossref","unstructured":"Long J, Shelhamer E, Darrell T (2015) Fully convolutional networks for semantic segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 3431\u20133440","DOI":"10.1109\/CVPR.2015.7298965"},{"key":"6337_CR24","first-page":"12077","volume":"34","author":"E Xie","year":"2021","unstructured":"Xie E, Wang W, Yu Z, Anandkumar A, Alvarez JM, Luo P (2021) Segformer: Simple and efficient design for semantic segmentation with transformers. Adv Neural Inf Process Syst 34:12077\u201312090","journal-title":"Adv Neural Inf Process Syst"},{"key":"6337_CR25","doi-asserted-by":"crossref","unstructured":"Liu Z, Lin Y, Cao Y, Hu H, Wei Y, Zhang Z, Lin S, Guo B (2021) Swin transformer: Hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 10012\u201310022","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"6337_CR26","doi-asserted-by":"crossref","unstructured":"Russakovsky O, Deng J, Su H, Krause J, Satheesh S, Ma S, Huang Z, Karpathy A, Khosla A, Bernstein M, et al. (2015) Imagenet large scale visual recognition challenge. Int J Comput Vis 115:211\u2013252","DOI":"10.1007\/s11263-015-0816-y"},{"key":"6337_CR27","doi-asserted-by":"crossref","unstructured":"Gupta S, Girshick R, Arbel\u00e1ez P, Malik J (2014) Learning rich features from rgb-d images for object detection and segmentation. In: Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, 6-12 September, 2014, Proceedings, Part VII 13, pp 345\u2013360. Springer","DOI":"10.1007\/978-3-319-10584-0_23"},{"key":"6337_CR28","doi-asserted-by":"crossref","unstructured":"Li Z, Gan Y, Liang X, Yu Y, Cheng H, Lin L (2016) Lstm-cf: Unifying context modeling and fusion with lstms for rgb-d scene labeling. In: Computer Vision\u2013ECCV 2016: 14th European Conference, Amsterdam, The Netherlands, 11-14 October, 2016, Proceedings, Part II 14, pp 541\u2013557 . Springer","DOI":"10.1007\/978-3-319-46475-6_34"},{"key":"6337_CR29","doi-asserted-by":"crossref","unstructured":"Cheng Y, Cai R, Li Z, Zhao X, Huang K (2017) Locality-sensitive deconvolution networks with gated fusion for rgb-d indoor semantic segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 3029\u20133037","DOI":"10.1109\/CVPR.2017.161"},{"key":"6337_CR30","doi-asserted-by":"crossref","unstructured":"Liu S, Huang D, et al. (2018) Receptive field block net for accurate and fast object detection. In: Proceedings of the European Conference on Computer Vision (ECCV), pp 385\u2013400","DOI":"10.1007\/978-3-030-01252-6_24"},{"key":"6337_CR31","doi-asserted-by":"crossref","unstructured":"Lin T-Y, Doll\u00e1r P, Girshick R, He K, Hariharan B, Belongie S (2017) Feature pyramid networks for object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 2117\u20132125","DOI":"10.1109\/CVPR.2017.106"},{"issue":"9","key":"6337_CR32","doi-asserted-by":"publisher","first-page":"1904","DOI":"10.1109\/TPAMI.2015.2389824","volume":"37","author":"K He","year":"2015","unstructured":"He K, Zhang X, Ren S, Sun J (2015) Spatial pyramid pooling in deep convolutional networks for visual recognition. IEEE Trans Pattern Anal Mach Intell 37(9):1904\u20131916","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"6337_CR33","doi-asserted-by":"crossref","unstructured":"Chen L-C, Zhu Y, Papandreou G, Schroff F, Adam H (2018) Encoder-decoder with atrous separable convolution for semantic image segmentation. In: Proceedings of the European Conference on Computer Vision (ECCV), pp 801\u2013818","DOI":"10.1007\/978-3-030-01234-2_49"},{"issue":"4","key":"6337_CR34","doi-asserted-by":"publisher","first-page":"834","DOI":"10.1109\/TPAMI.2017.2699184","volume":"40","author":"L-C Chen","year":"2017","unstructured":"Chen L-C, Papandreou G, Kokkinos I, Murphy K, Yuille AL (2017) Deeplab: Semantic image segmentation with deep convolutional nets, atrous convolution, and fully connected crfs. IEEE Trans Pattern Anal Mach Intell 40(4):834\u2013848","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"6337_CR35","unstructured":"Kingma DP, Ba J (2014) Adam: A method for stochastic optimization. arXiv:1412.6980"},{"key":"6337_CR36","doi-asserted-by":"crossref","unstructured":"Song S, Lichtenberg SP, Xiao J (2015) Sun rgb-d: A rgb-d scene understanding benchmark suite. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 567\u2013576","DOI":"10.1109\/CVPR.2015.7298655"},{"key":"6337_CR37","doi-asserted-by":"publisher","first-page":"658","DOI":"10.1109\/LSP.2021.3066071","volume":"28","author":"G Zhang","year":"2021","unstructured":"Zhang G, Xue J-H, Xie P, Yang S, Wang G (2021) Non-local aggregation for rgb-d semantic segmentation. IEEE Signal Process Lett 28:658\u2013662","journal-title":"IEEE Signal Process Lett"},{"key":"6337_CR38","doi-asserted-by":"crossref","unstructured":"Ye H, Xu D (2022) Inverted pyramid multi-task transformer for dense scene understanding. In: European Conference on Computer Vision, pp 514\u2013530 . Springer","DOI":"10.1007\/978-3-031-19812-0_30"},{"key":"6337_CR39","doi-asserted-by":"crossref","unstructured":"Girdhar R, Singh M, Ravi N, Maaten L, Joulin A, Misra I (2022) Omnivore: A single model for many visual modalities. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 16102\u201316112","DOI":"10.1109\/CVPR52688.2022.01563"},{"key":"6337_CR40","unstructured":"Pang Y, Zhao X, Zhang L, Lu H (2023) Comptr: Towards diverse bi-source dense prediction tasks via a simple yet general complementary transformer. arXiv:2307.12349"},{"key":"6337_CR41","doi-asserted-by":"crossref","unstructured":"Xing Y, Wang J, Zeng G (2020) Malleable 2.5 d convolution: Learning receptive fields along the depth-axis for rgb-d scene parsing. In: European Conference on Computer Vision, pp 555\u2013571 . Springer","DOI":"10.1007\/978-3-030-58529-7_33"},{"key":"6337_CR42","doi-asserted-by":"crossref","unstructured":"Chen X, Lin K-Y, Wang J, Wu W, Qian C, Li H, Zeng G (2020) Bi-directional cross-modality feature propagation with separation-and-aggregation gate for rgb-d semantic segmentation. In: European Conference on Computer Vision, pp 561\u2013577 . Springer","DOI":"10.1007\/978-3-030-58621-8_33"},{"key":"6337_CR43","doi-asserted-by":"crossref","unstructured":"Borse S, Wang Y, Zhang Y, Porikli F (2021) Inverseform: A loss function for structured boundary-aware segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 5901\u20135911","DOI":"10.1109\/CVPR46437.2021.00584"},{"key":"6337_CR44","doi-asserted-by":"crossref","unstructured":"Qi X, Liao R, Jia J, Fidler S, Urtasun R (2017) 3d graph neural networks for rgbd semantic segmentation. In: Proceedings of the IEEE International Conference on Computer Vision, pp 5199\u20135208","DOI":"10.1109\/ICCV.2017.556"},{"key":"6337_CR45","doi-asserted-by":"crossref","unstructured":"Lin D, Chen G, Cohen-Or D, Heng P-A, Huang H (2017) Cascaded feature network for semantic segmentation of rgb-d images. In: Proceedings of the IEEE International Conference on Computer Vision, pp 1311\u20131319","DOI":"10.1109\/ICCV.2017.147"},{"key":"6337_CR46","doi-asserted-by":"crossref","unstructured":"Seichter D, K\u00f6hler M, Lewandowski B, Wengefeld T, Gross H-M (2021) Efficient rgb-d semantic segmentation for indoor scene analysis. In: 2021 IEEE International Conference on Robotics and Automation (ICRA), pp 13525\u201313531 . IEEE","DOI":"10.1109\/ICRA48506.2021.9561675"},{"key":"6337_CR47","doi-asserted-by":"crossref","unstructured":"Zhou W, Yang E, Lei J, Wan J, Yu L (2022) Pgdenet: Progressive guided fusion and depth enhancement network for rgb-d indoor scene parsing. IEEE Trans Multimed, 1\u20131","DOI":"10.1109\/TMM.2022.3161852"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-025-06337-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-025-06337-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-025-06337-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,19]],"date-time":"2025-09-19T19:30:30Z","timestamp":1758310230000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-025-06337-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,2,15]]},"references-count":47,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2025,5]]}},"alternative-id":["6337"],"URL":"https:\/\/doi.org\/10.1007\/s10489-025-06337-0","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"value":"0924-669X","type":"print"},{"value":"1573-7497","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,2,15]]},"assertion":[{"value":"4 February 2025","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 February 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of Interest"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics Approval"}},{"value":"Not applicable.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to Participate"}},{"value":"Not applicable.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for Publication"}}],"article-number":"454"}}