{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,29]],"date-time":"2026-01-29T23:38:53Z","timestamp":1769729933719,"version":"3.49.0"},"reference-count":67,"publisher":"Springer Science and Business Media LLC","license":[{"start":{"date-parts":[[2024,5,9]],"date-time":"2024-05-09T00:00:00Z","timestamp":1715212800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,5,9]],"date-time":"2024-05-09T00:00:00Z","timestamp":1715212800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61901392"],"award-info":[{"award-number":["61901392"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100004829","name":"Department of Science and Technology of Sichuan Province","doi-asserted-by":"publisher","award":["2021YJ0109"],"award-info":[{"award-number":["2021YJ0109"]}],"id":[{"id":"10.13039\/501100004829","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"DOI":"10.1007\/s11042-024-19051-9","type":"journal-article","created":{"date-parts":[[2024,5,9]],"date-time":"2024-05-09T06:43:55Z","timestamp":1715237035000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["CLGFormer: Cross-Level-Guided transformer for RGB-D semantic segmentation"],"prefix":"10.1007","author":[{"given":"Tao","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qunbing","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dandan","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mingming","family":"Sun","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ting","family":"Hu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,5,9]]},"reference":[{"issue":"4","key":"19051_CR1","doi-asserted-by":"publisher","first-page":"2136","DOI":"10.1016\/j.eswa.2014.09.043","volume":"42","author":"HVH Ayala","year":"2015","unstructured":"Ayala HVH, dos Santos FM, Mariani VC et al (2015) Image thresholding segmentation based on a novel beta differential evolution approach. Expert Syst Appl 42(4):2136\u20132142. https:\/\/doi.org\/10.1016\/j.eswa.2014.09.043","journal-title":"Expert Syst Appl"},{"key":"19051_CR2","doi-asserted-by":"publisher","unstructured":"Long J, Shelhamer E, Darrell T (2015) Fully convolutional networks for semantic segmentation. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 3431\u20133440. https:\/\/doi.org\/10.1109\/CVPR.2015.7298965","DOI":"10.1109\/CVPR.2015.7298965"},{"issue":"12","key":"19051_CR3","doi-asserted-by":"publisher","first-page":"2481","DOI":"10.1109\/TPAMI.2016.2644615","volume":"39","author":"V Badrinarayanan","year":"2017","unstructured":"Badrinarayanan V, Kendall A, Cipolla R (2017) Segnet: A deep convolutional encoder-decoder architecture for image segmentation. IEEE Trans Pattern Anal Mach Intell 39(12):2481\u20132495. https:\/\/doi.org\/10.1109\/TPAMI.2016.2644615","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"19051_CR4","doi-asserted-by":"publisher","unstructured":"Lin G, Milan A, Shen C, et\u00a0al (2017) Refinenet: Multi-path refinement networks for high-resolution semantic segmentation. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 1925\u20131934. https:\/\/doi.org\/10.1109\/CVPR.2017.549","DOI":"10.1109\/CVPR.2017.549"},{"key":"19051_CR5","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2023.120200","volume":"227","author":"AN Tabata","year":"2023","unstructured":"Tabata AN, Zimmer A, dos Santos Coelho L et al (2023) Analyzing carla\u2019s performance for 2d object detection and monocular depth estimation based on deep learning approaches. Expert Syst Appl 227:120200. https:\/\/doi.org\/10.1016\/j.eswa.2023.120200","journal-title":"Expert Syst Appl"},{"key":"19051_CR6","doi-asserted-by":"publisher","DOI":"10.1016\/j.ins.2023.119701","volume":"651","author":"Y Zheng","year":"2023","unstructured":"Zheng Y, Demetrio L, Cin(\u00e0) AE, et al (2023) Hardening rgb-d object recognition systems against adversarial patch attacks. Inf Sci 651:119701. https:\/\/doi.org\/10.1016\/j.ins.2023.119701","journal-title":"Inf Sci"},{"key":"19051_CR7","doi-asserted-by":"publisher","unstructured":"Hazirbas C, Ma L, Domokos C et\u00a0al (2017) Fusenet: Incorporating depth into semantic segmentation via fusion-based cnn architecture. In: Computer Vision\u2013ACCV 2016: 13th Asian Conference on Computer Vision, Taipei, Taiwan, November 20-24, 2016, Revised Selected Papers, Part I 13, Springer, pp 213\u2013228. https:\/\/doi.org\/10.1007\/978-3-319-54181-5_14","DOI":"10.1007\/978-3-319-54181-5_14"},{"key":"19051_CR8","doi-asserted-by":"publisher","unstructured":"Jiang J, Zheng L, Luo F et\u00a0al (2018) Rednet: Residual encoder-decoder network for indoor rgb-d semantic segmentation. arXiv:1806.01054. https:\/\/doi.org\/10.48550\/arXiv.1806.01054","DOI":"10.48550\/arXiv.1806.01054"},{"key":"19051_CR9","doi-asserted-by":"publisher","unstructured":"Seichter D, K\u00f6hler M, Lewandowski B et\u00a0al (2021) Efficient rgb-d semantic segmentation for indoor scene analysis. In: 2021 IEEE International Conference on Robotics and Automation (ICRA), pp 13525\u201313531. https:\/\/doi.org\/10.1109\/ICRA48506.2021.9561675","DOI":"10.1109\/ICRA48506.2021.9561675"},{"issue":"4","key":"19051_CR10","doi-asserted-by":"publisher","first-page":"5558","DOI":"10.1109\/LRA.2020.3007457","volume":"5","author":"L Sun","year":"2020","unstructured":"Sun L, Yang K, Hu X et al (2020) Real-time fusion network for rgb-d semantic segmentation incorporating unexpected obstacle detection for road-driving images. IEEE Robot Autom Lett 5(4):5558\u20135565. https:\/\/doi.org\/10.1109\/LRA.2020.3007457","journal-title":"IEEE Robot Autom Lett"},{"key":"19051_CR11","doi-asserted-by":"publisher","DOI":"10.1109\/JSEN.2023.3304637","author":"Y Zhang","year":"2023","unstructured":"Zhang Y, Xiong C, Liu J et al (2023) Spatial-information guided adaptive context-aware network for efficient rgb-d semantic segmentation. IEEE Sensors J. https:\/\/doi.org\/10.1109\/JSEN.2023.3304637","journal-title":"IEEE Sensors J"},{"issue":"12","key":"19051_CR12","doi-asserted-by":"publisher","first-page":"14679","DOI":"10.1109\/TITS.2023.3300537","volume":"24","author":"J Zhang","year":"2023","unstructured":"Zhang J, Liu H, Yang K et al (2023) Cmx: Cross-modal fusion for rgb-x semantic segmentation with transformers. IEEE Trans Intell Transp Syst 24(12):14679\u201314694. https:\/\/doi.org\/10.1109\/TITS.2023.3300537","journal-title":"IEEE Trans Intell Transp Syst"},{"issue":"1","key":"19051_CR13","doi-asserted-by":"publisher","first-page":"20305","DOI":"10.1038\/s41598-022-24836-9","volume":"12","author":"S Jiang","year":"2022","unstructured":"Jiang S, Xu Y, Li D et al (2022) Multi-scale fusion for rgb-d indoor semantic segmentation. Sci Rep 12(1):20305. https:\/\/doi.org\/10.1038\/s41598-022-24836-9","journal-title":"Sci Rep"},{"issue":"7","key":"19051_CR14","doi-asserted-by":"publisher","first-page":"4486","DOI":"10.1109\/TCSVT.2021.3127149","volume":"32","author":"Z Liu","year":"2021","unstructured":"Liu Z, Tan Y, He Q et al (2021) Swinnet: Swin transformer drives edge-aware rgb-d and rgb-t salient object detection. IEEE Trans Circ Syst Video Technol 32(7):4486\u20134497. https:\/\/doi.org\/10.1109\/TCSVT.2021.3127149","journal-title":"IEEE Trans Circ Syst Video Technol"},{"key":"19051_CR15","doi-asserted-by":"publisher","unstructured":"Wu Z, Zhou Z, Allibert G et al (2022) Transformer fusion for indoor rgb-d semantic segmentation. Available at SSRN 4251286. https:\/\/doi.org\/10.2139\/ssrn.4251286","DOI":"10.2139\/ssrn.4251286"},{"key":"19051_CR16","unstructured":"Dosovitskiy A, Beyer L, Kolesnikov A et\u00a0al (2021) An image is worth 16x16 words: Transformers for image recognition at scale. In: International conference on learning representations"},{"key":"19051_CR17","doi-asserted-by":"publisher","unstructured":"Zheng S, Lu J, Zhao H et\u00a0al (2021) Rethinking semantic segmentation from a sequence-to-sequence perspective with transformers. In: 2021 IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp 6877\u20136886. https:\/\/doi.org\/10.1109\/CVPR46437.2021.00681","DOI":"10.1109\/CVPR46437.2021.00681"},{"key":"19051_CR18","doi-asserted-by":"publisher","unstructured":"Xu J, Shi W, Gao P et\u00a0al (2022) Uperformer: A multi-scale transformer-based decoder for semantic segmentation. arXiv:2211.13928. https:\/\/doi.org\/10.48550\/arXiv.2211.13928","DOI":"10.48550\/arXiv.2211.13928"},{"key":"19051_CR19","doi-asserted-by":"crossref","unstructured":"Chen J, Lu Y, Yu Q et\u00a0al (2021) Transunet: Transformers make strong encoders for medical image segmentation. arXiv:2102.04306","DOI":"10.1109\/IGARSS46834.2022.9883628"},{"key":"19051_CR20","doi-asserted-by":"publisher","unstructured":"Wang H, Cao P, Wang J et\u00a0al (2022) Uctransnet: rethinking the skip connections in u-net from a channel-wise perspective with transformer. In: Proceedings of the AAAI conference on artificial intelligence, pp 2441\u20132449. https:\/\/doi.org\/10.1609\/aaai.v36i3.20144","DOI":"10.1609\/aaai.v36i3.20144"},{"key":"19051_CR21","doi-asserted-by":"publisher","unstructured":"Sanida T, Sideris A, Dasygenis M (2020) A heterogeneous implementation of the sobel edge detection filter using opencl. In: 2020 9th International conference on modern circuits and systems technologies (MOCAST), pp 1\u20134. https:\/\/doi.org\/10.1109\/MOCAST49295.2020.9200249","DOI":"10.1109\/MOCAST49295.2020.9200249"},{"key":"19051_CR22","doi-asserted-by":"publisher","unstructured":"Silberman N, Hoiem D, Kohli P et\u00a0al (2012) Indoor segmentation and support inference from rgbd images. In: Computer Vision\u2013ECCV 2012: 12th European Conference on Computer Vision, Florence, Italy, October 7-13, 2012, Proceedings, Part V 12, Springer, pp 746\u2013760. https:\/\/doi.org\/10.1007\/978-3-642-33715-4_54","DOI":"10.1007\/978-3-642-33715-4_54"},{"key":"19051_CR23","doi-asserted-by":"publisher","unstructured":"Cordts M, Omran M, Ramos S et\u00a0al (2016) The cityscapes dataset for semantic urban scene understanding. In: 2016 IEEE Conference on computer vision and pattern recognition (CVPR), pp 3213\u20133223. https:\/\/doi.org\/10.1109\/CVPR.2016.350","DOI":"10.1109\/CVPR.2016.350"},{"key":"19051_CR24","doi-asserted-by":"publisher","first-page":"961","DOI":"10.1007\/s11263-018-1070-x","volume":"126","author":"H Abu Alhaija","year":"2018","unstructured":"Abu Alhaija H, Mustikovela SK, Mescheder L et al (2018) Augmented reality meets computer vision: Efficient data generation for urban driving scenes. Int J Comput Vis 126:961\u2013972. https:\/\/doi.org\/10.1007\/s11263-018-1070-x","journal-title":"Int J Comput Vis"},{"key":"19051_CR25","doi-asserted-by":"publisher","unstructured":"Lee S, Park SJ, Hong KS (2017) Rdfnet: Rgb-d multi-level residual feature fusion for indoor semantic segmentation. In: 2017 IEEE International conference on computer vision (ICCV), pp 4990\u20134999. https:\/\/doi.org\/10.1109\/ICCV.2017.533","DOI":"10.1109\/ICCV.2017.533"},{"key":"19051_CR26","doi-asserted-by":"publisher","unstructured":"Chen X, Lin KY, Wang J et\u00a0al (2020) Bi-directional cross-modality feature propagation with separation-and-aggregation gate for rgb-d semantic segmentation. In: European Conference on Computer Vision, Springer, pp 561\u2013577. https:\/\/doi.org\/10.1007\/978-3-030-58621-8_33","DOI":"10.1007\/978-3-030-58621-8_33"},{"issue":"18","key":"19051_CR27","doi-asserted-by":"publisher","first-page":"3943","DOI":"10.3390\/electronics12183943","volume":"12","author":"X Xu","year":"2023","unstructured":"Xu X, Liu J, Liu H (2023) Interactive efficient multi-task network for rgb-d semantic segmentation. Electronics 12(18):3943. https:\/\/doi.org\/10.3390\/electronics12183943","journal-title":"Electronics"},{"issue":"25","key":"19051_CR28","doi-asserted-by":"publisher","first-page":"35815","DOI":"10.1007\/s11042-021-11395-w","volume":"81","author":"W Zou","year":"2022","unstructured":"Zou W, Peng Y, Zhang Z et al (2022) Rgb-d gate-guided edge distillation for indoor semantic segmentation. Multimed Tools Appl 81(25):35815\u201335830. https:\/\/doi.org\/10.1007\/s11042-021-11395-w","journal-title":"Multimed Tools Appl"},{"key":"19051_CR29","doi-asserted-by":"publisher","DOI":"10.1016\/j.engappai.2023.106885","volume":"126","author":"Y Pan","year":"2023","unstructured":"Pan Y, Zhou W, Qian X et al (2023) Cginet: Cross-modality grade interaction network for rgb-t crowd counting. Eng Appl Artif Intell 126:106885. https:\/\/doi.org\/10.1016\/j.engappai.2023.106885","journal-title":"Eng Appl Artif Intell"},{"key":"19051_CR30","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2021.108468","volume":"124","author":"H Zhou","year":"2022","unstructured":"Zhou H, Qi L, Huang H et al (2022) Canet: Co-attention network for rgb-d semantic segmentation. Pattern Recog 124:108468. https:\/\/doi.org\/10.1016\/j.patcog.2021.108468","journal-title":"Pattern Recog"},{"key":"19051_CR31","doi-asserted-by":"publisher","unstructured":"Fu J, Liu J, Tian H et\u00a0al (2019) Dual attention network for scene segmentation. In: 2019 IEEE\/CVF Conference on computer vision and pattern recognition (CVPR), pp 3141\u20133149. https:\/\/doi.org\/10.1109\/CVPR.2019.00326","DOI":"10.1109\/CVPR.2019.00326"},{"key":"19051_CR32","doi-asserted-by":"publisher","unstructured":"Hu X, Yang K, Fei L et\u00a0al (2019) Acnet: Attention based network to exploit complementary features for rgbd semantic segmentation. In: 2019 IEEE international conference on image processing (ICIP), pp 1440\u20131444. https:\/\/doi.org\/10.1109\/ICIP.2019.8803025","DOI":"10.1109\/ICIP.2019.8803025"},{"key":"19051_CR33","doi-asserted-by":"publisher","unstructured":"Zhang Y, Yang Y, Xiong C et\u00a0al (2022) Attention-based dual supervised decoder for rgbd semantic segmentation. arXiv:2201.01427. https:\/\/doi.org\/10.48550\/arXiv.2201.01427","DOI":"10.48550\/arXiv.2201.01427"},{"key":"19051_CR34","doi-asserted-by":"publisher","unstructured":"Seichter D, Fischedick SB, K\u00f6hler M et\u00a0al (2022) Efficient multi-task rgb-d scene analysis for indoor environments. In: 2022 International joint conference on neural networks (IJCNN), pp 1\u201310. https:\/\/doi.org\/10.1109\/IJCNN55064.2022.9892852","DOI":"10.1109\/IJCNN55064.2022.9892852"},{"key":"19051_CR35","first-page":"12077","volume":"34","author":"E Xie","year":"2021","unstructured":"Xie E, Wang W, Yu Z et al (2021) Segformer: Simple and efficient design for semantic segmentation with transformers. Adv Neural Inf Process Syst 34:12077\u201312090","journal-title":"Adv Neural Inf Process Syst"},{"key":"19051_CR36","doi-asserted-by":"publisher","unstructured":"Wang W, Xie E, Li X et\u00a0al (2021) Pyramid vision transformer: A versatile backbone for dense prediction without convolutions. In: 2021 IEEE\/CVF International conference on computer vision (ICCV), pp 548\u2013558. https:\/\/doi.org\/10.1109\/ICCV48922.2021.00061","DOI":"10.1109\/ICCV48922.2021.00061"},{"key":"19051_CR37","doi-asserted-by":"publisher","unstructured":"Wu H, Xiao B, Codella N et\u00a0al (2021) Cvt: Introducing convolutions to vision transformers. In: 2021 IEEE\/CVF International conference on computer vision (ICCV), pp 22\u201331. https:\/\/doi.org\/10.1109\/ICCV48922.2021.00009","DOI":"10.1109\/ICCV48922.2021.00009"},{"key":"19051_CR38","doi-asserted-by":"publisher","unstructured":"Wang Y, Chen X, Cao L et\u00a0al (2022) Multimodal token fusion for vision transformers. In: 2022 IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp 12176\u201312185. https:\/\/doi.org\/10.1109\/CVPR52688.2022.01187","DOI":"10.1109\/CVPR52688.2022.01187"},{"key":"19051_CR39","doi-asserted-by":"publisher","unstructured":"Liu Z, Lin Y, Cao Y et\u00a0al (2021) Swin transformer: Hierarchical vision transformer using shifted windows. In: 2021 IEEE\/CVF international conference on computer vision (ICCV), pp 9992\u201310002. https:\/\/doi.org\/10.1109\/ICCV48922.2021.00986","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"19051_CR40","doi-asserted-by":"publisher","unstructured":"Ying X, Chuah MC (2022) Uctnet: Uncertainty-aware cross-modal transformer network for indoor rgb-d semantic segmentation. In: European Conference on Computer Vision, Springer, pp 20\u201337. https:\/\/doi.org\/10.1007\/978-3-031-20056-4_2","DOI":"10.1007\/978-3-031-20056-4_2"},{"key":"19051_CR41","doi-asserted-by":"publisher","unstructured":"He K, Zhang X, Ren S et\u00a0al (2016) Deep residual learning for image recognition. In: 2016 IEEE Conference on computer vision and pattern recognition (CVPR), pp 770\u2013778. https:\/\/doi.org\/10.1109\/CVPR.2016.90","DOI":"10.1109\/CVPR.2016.90"},{"key":"19051_CR42","doi-asserted-by":"publisher","unstructured":"Hu J, Shen L, Sun G (2018) Squeeze-and-excitation networks. In: 2018 IEEE\/CVF conference on computer vision and pattern recognition, pp 7132\u20137141. https:\/\/doi.org\/10.1109\/CVPR.2018.00745","DOI":"10.1109\/CVPR.2018.00745"},{"key":"19051_CR43","unstructured":"Lee CY, Xie S, Gallagher P et\u00a0al (2015) Deeply-supervised nets. In: Artificial intelligence and statistics, Pmlr, pp 562\u2013570"},{"key":"19051_CR44","unstructured":"Kingma DP, Ba J (2015) Adam: A method for stochastic optimization. In: Bengio Y, LeCun Y (eds) 3rd International Conference on Learning Representations, ICLR 2015, San Diego, CA, USA, May 7-9, 2015, Conference Track Proceedings"},{"issue":"21","key":"19051_CR45","doi-asserted-by":"publisher","first-page":"8520","DOI":"10.3390\/s22218520","volume":"22","author":"L Zhu","year":"2022","unstructured":"Zhu L, Kang Z, Zhou M et al (2022) Cmanet: Cross-modality attention network for indoor-scene semantic segmentation. Sensors 22(21):8520. https:\/\/doi.org\/10.3390\/s22218520","journal-title":"Sensors"},{"key":"19051_CR46","doi-asserted-by":"publisher","unstructured":"Xu Y, Li X, Yuan H et\u00a0al (2023) Multi-task learning with multi-query transformer for dense prediction. IEEE Trans Circ Syst Video Technol pp 1\u20131. https:\/\/doi.org\/10.1109\/TCSVT.2023.3292995","DOI":"10.1109\/TCSVT.2023.3292995"},{"key":"19051_CR47","doi-asserted-by":"publisher","first-page":"2313","DOI":"10.1109\/TIP.2021.3049332","volume":"30","author":"LZ Chen","year":"2021","unstructured":"Chen LZ, Lin Z, Wang Z et al (2021) Spatial information guided convolution for real-time rgbd semantic segmentation. IEEE Trans Image Process 30:2313\u20132324. https:\/\/doi.org\/10.1109\/TIP.2021.3049332","journal-title":"IEEE Trans Image Process"},{"key":"19051_CR48","doi-asserted-by":"publisher","unstructured":"Yang Y, Xu Y, Zhang C et\u00a0al (2022) Hierarchical vision transformer with channel attention for rgb-d image segmentation. In: Proceedings of the 4th international symposium on signal processing systems, pp 68\u201373. https:\/\/doi.org\/10.1145\/3532342.3532352","DOI":"10.1145\/3532342.3532352"},{"key":"19051_CR49","doi-asserted-by":"publisher","unstructured":"Xing Y, Wang J, Zeng G (2020) Malleable 2.5 d convolution: Learning receptive fields along the depth-axis for rgb-d scene parsing. In: European conference on computer vision, Springer, pp 555\u2013571. https:\/\/doi.org\/10.1007\/978-3-030-58529-7_33","DOI":"10.1007\/978-3-030-58529-7_33"},{"key":"19051_CR50","doi-asserted-by":"publisher","unstructured":"Cao J, Leng H, Lischinski D et\u00a0al (2021) Shapeconv: Shape-aware convolutional layer for indoor rgb-d semantic segmentation. In: 2021 IEEE\/CVF International conference on computer vision (ICCV), pp 7068\u20137077. https:\/\/doi.org\/10.1109\/ICCV48922.2021.00700","DOI":"10.1109\/ICCV48922.2021.00700"},{"key":"19051_CR51","doi-asserted-by":"publisher","first-page":"2503","DOI":"10.1109\/TMM.2022.3147664","volume":"25","author":"X Zhang","year":"2023","unstructured":"Zhang X, Zhang S, Cui Z et al (2023) Tube-embedded transformer for pixel prediction. IEEE Trans Multimed 25:2503\u20132514. https:\/\/doi.org\/10.1109\/TMM.2022.3147664","journal-title":"IEEE Trans Multimed"},{"key":"19051_CR52","doi-asserted-by":"publisher","unstructured":"Zhu X, Wang X, Freer J et\u00a0al (2023) Clothes grasping and unfolding based on rgb-d semantic segmentation. In: 2023 IEEE International conference on robotics and automation (ICRA), pp 9471\u20139477. https:\/\/doi.org\/10.1109\/ICRA48891.2023.10160268","DOI":"10.1109\/ICRA48891.2023.10160268"},{"key":"19051_CR53","doi-asserted-by":"publisher","unstructured":"Cheng Y, Cai R, Li Z et\u00a0al (2017) Locality-sensitive deconvolution networks with gated fusion for rgb-d indoor semantic segmentation. In: 2017 IEEE Conference on computer vision and pattern recognition (CVPR), pp 1475\u20131483. https:\/\/doi.org\/10.1109\/CVPR.2017.161","DOI":"10.1109\/CVPR.2017.161"},{"key":"19051_CR54","doi-asserted-by":"publisher","unstructured":"Xiong Z, Yuan Y, Guo N et\u00a0al (2020) Variational context-deformable convnets for indoor scene parsing. In: 2020 IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp 3991\u20134001. https:\/\/doi.org\/10.1109\/CVPR42600.2020.00405","DOI":"10.1109\/CVPR42600.2020.00405"},{"key":"19051_CR55","doi-asserted-by":"publisher","unstructured":"Orsic M, Kreso I, Bevandic P et\u00a0al (2019) In defense of pre-trained imagenet architectures for real-time semantic segmentation of road-driving images. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 12607\u201312616. https:\/\/doi.org\/10.1109\/CVPR.2019.01289","DOI":"10.1109\/CVPR.2019.01289"},{"key":"19051_CR56","doi-asserted-by":"publisher","unstructured":"Hung SW, Lo SY, Hang HM (2019) Incorporating luminance, depth and color information by a fusion-based network for semantic segmentation. In: 2019 IEEE International conference on image processing (ICIP), IEEE, pp 2374\u20132378. https:\/\/doi.org\/10.1109\/ICIP.2019.8803360","DOI":"10.1109\/ICIP.2019.8803360"},{"issue":"4","key":"19051_CR57","doi-asserted-by":"publisher","first-page":"5558","DOI":"10.1109\/LRA.2020.3007457","volume":"5","author":"L Sun","year":"2020","unstructured":"Sun L, Yang K, Hu X et al (2020) Real-time fusion network for rgb-d semantic segmentation incorporating unexpected obstacle detection for road-driving images. IEEE Robot Autom Lett 5(4):5558\u20135565. https:\/\/doi.org\/10.1109\/LRA.2020.3007457","journal-title":"IEEE Robot Autom Lett"},{"key":"19051_CR58","doi-asserted-by":"publisher","unstructured":"Xu D, Ouyang W, Wang X et\u00a0al (2018) Pad-net: Multi-tasks guided prediction-and-distillation network for simultaneous depth estimation and scene parsing. In: 2018 IEEE\/CVF Conference on computer vision and pattern recognition, pp 675\u2013684. https:\/\/doi.org\/10.1109\/CVPR.2018.00077","DOI":"10.1109\/CVPR.2018.00077"},{"key":"19051_CR59","doi-asserted-by":"publisher","unstructured":"Chen LC, Zhu Y, Papandreou G et\u00a0al (2018) Encoder-decoder with atrous separable convolution for semantic image segmentation. In: Proceedings of the European conference on computer vision (ECCV), pp 801\u2013818. https:\/\/doi.org\/10.1007\/978-3-030-01234-2_49","DOI":"10.1007\/978-3-030-01234-2_49"},{"issue":"17","key":"19051_CR60","doi-asserted-by":"publisher","first-page":"9924","DOI":"10.3390\/app13179924","volume":"13","author":"S Chen","year":"2023","unstructured":"Chen S, Tang M, Dong R et al (2023) Encoder-decoder structure fusing depth information for outdoor semantic segmentation. Appl Sci 13(17):9924","journal-title":"Appl Sci"},{"key":"19051_CR61","doi-asserted-by":"publisher","unstructured":"Kong S, Fowlkes C (2018) Recurrent scene parsing with perspective understanding in the loop. In: 2018 IEEE\/CVF conference on computer vision and pattern recognition, pp 956\u2013965. https:\/\/doi.org\/10.1109\/CVPR.2018.00106","DOI":"10.1109\/CVPR.2018.00106"},{"key":"19051_CR62","doi-asserted-by":"publisher","DOI":"10.1109\/TIM.2023.3328708","author":"L Sun","year":"2023","unstructured":"Sun L, Bockman J, Sun C (2023) A framework for leveraging inter-image information in stereo images for enhanced semantic segmentation in autonomous driving. IEEE Trans Instrum Meas. https:\/\/doi.org\/10.1109\/TIM.2023.3328708","journal-title":"IEEE Trans Instrum Meas"},{"key":"19051_CR63","doi-asserted-by":"publisher","unstructured":"Kong S, Fowlkes C (2018) Pixel-wise attentional gating for parsimonious pixel labeling. arXiv:1805.01556. https:\/\/doi.org\/10.48550\/arXiv.1805.01556","DOI":"10.48550\/arXiv.1805.01556"},{"key":"19051_CR64","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2023.109557","volume":"140","author":"T Singha","year":"2023","unstructured":"Singha T, Pham DS, Krishna A (2023) A real-time semantic segmentation model using iteratively shared features in multiple sub-encoders. Pattern Recog 140:109557. https:\/\/doi.org\/10.1016\/j.patcog.2023.109557","journal-title":"Pattern Recog"},{"key":"19051_CR65","doi-asserted-by":"publisher","unstructured":"Ochs M, Kretz A, Mester R (2019) Sdnet: Semantically guided depth estimation network. In: Pattern Recognition: 41st DAGM German Conference, DAGM GCPR 2019, Dortmund, Germany, September 10\u201313, 2019, Proceedings 41, Springer, pp 288\u2013302. https:\/\/doi.org\/10.1007\/978-3-030-33676-9_20","DOI":"10.1007\/978-3-030-33676-9_20"},{"key":"19051_CR66","doi-asserted-by":"publisher","unstructured":"Singha T, Pham DS, Krishna A (2022) Sdbnet: Lightweight real-time semantic segmentation using short-term dense bottleneck. In: 2022 International Conference on Digital Image Computing: Techniques and Applications (DICTA), pp 1\u20138. https:\/\/doi.org\/10.1109\/DICTA56598.2022.10034634","DOI":"10.1109\/DICTA56598.2022.10034634"},{"key":"19051_CR67","doi-asserted-by":"publisher","unstructured":"Klingner M, Term\u00f6hlen JA, Mikolajczyk J et\u00a0al (2020) Self-supervised monocular depth estimation: Solving the dynamic object problem by semantic guidance. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XX 16, Springer, pp 582\u2013600. https:\/\/doi.org\/10.1007\/978-3-030-58565-5_35","DOI":"10.1007\/978-3-030-58565-5_35"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-024-19051-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-024-19051-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-024-19051-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,5,9]],"date-time":"2024-05-09T06:49:10Z","timestamp":1715237350000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-024-19051-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,5,9]]},"references-count":67,"alternative-id":["19051"],"URL":"https:\/\/doi.org\/10.1007\/s11042-024-19051-9","relation":{},"ISSN":["1573-7721"],"issn-type":[{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,5,9]]},"assertion":[{"value":"6 September 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 December 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 March 2024","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 May 2024","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Not applicable","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical and informed consent for data used"}},{"value":"The authors have no financial or proprietary interests in any material discussed in this article","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing Interests"}}]}}