{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T12:55:59Z","timestamp":1761396959085,"version":"3.37.3"},"reference-count":44,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2024,12,4]],"date-time":"2024-12-04T00:00:00Z","timestamp":1733270400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,4]],"date-time":"2024-12-04T00:00:00Z","timestamp":1733270400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Science Foundation of Zhejiang Province of China","award":["LY24F020010"],"award-info":[{"award-number":["LY24F020010"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62006209"],"award-info":[{"award-number":["62006209"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"Science Foundation of Zhejiang Sci-Tech University","award":["18022225-Y"],"award-info":[{"award-number":["18022225-Y"]}]},{"DOI":"10.13039\/501100013254","name":"National College Students Innovation and Entrepreneurship Training Program","doi-asserted-by":"publisher","award":["No.202310338013"],"award-info":[{"award-number":["No.202310338013"]}],"id":[{"id":"10.13039\/501100013254","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2025,1]]},"DOI":"10.1007\/s11227-024-06760-z","type":"journal-article","created":{"date-parts":[[2024,12,4]],"date-time":"2024-12-04T11:38:33Z","timestamp":1733312313000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["CDMANet: central difference mutual attention network for RGB-D semantic segmentation"],"prefix":"10.1007","volume":"81","author":[{"given":"Mengjiao","family":"Ge","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wen","family":"Su","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jinfeng","family":"Gao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guoqiang","family":"Jia","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,12,4]]},"reference":[{"key":"6760_CR1","doi-asserted-by":"publisher","first-page":"196","DOI":"10.1016\/j.isprsjprs.2022.06.008","volume":"190","author":"L Wang","year":"2021","unstructured":"Wang L, Li R, Zhang C, Fang S, Duan C, Meng X, Atkinson PM (2021) Unetformer: a unet-like transformer for efficient semantic segmentation of remote sensing urban scene imagery. ISPRS J Photogramm Remote Sens 190:196\u2013214","journal-title":"ISPRS J Photogramm Remote Sens"},{"key":"6760_CR2","doi-asserted-by":"publisher","unstructured":"Xu J, Zhang R, Dou J, Zhu Y, Sun J, Pu S (2021) Rpvnet: a deep and efficient range-point-voxel fusion network for lidar point cloud segmentation. In: 2021 IEEE\/CVF International Conference on Computer Vision (ICCV), pp 16004\u201316013. https:\/\/doi.org\/10.1109\/ICCV48922.2021.01572","DOI":"10.1109\/ICCV48922.2021.01572"},{"key":"6760_CR3","doi-asserted-by":"publisher","unstructured":"Basak H, Yin Z (2023) Pseudo-label guided contrastive learning for semi-supervised medical image segmentation. In: 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp 19786\u201319797. https:\/\/doi.org\/10.1109\/CVPR52729.2023.01895","DOI":"10.1109\/CVPR52729.2023.01895"},{"key":"6760_CR4","doi-asserted-by":"publisher","first-page":"2313","DOI":"10.1109\/TIP.2021.3049332","volume":"30","author":"L-Z Chen","year":"2021","unstructured":"Chen L-Z, Lin Z, Wang Z, Yang Y-L, Cheng M-M (2021) Spatial information guided convolution for real-time rgbd semantic segmentation. IEEE Trans Image Process 30:2313\u20132324. https:\/\/doi.org\/10.1109\/TIP.2021.3049332","journal-title":"IEEE Trans Image Process"},{"key":"6760_CR5","doi-asserted-by":"publisher","unstructured":"Lee S, Park S-J, Hong K-S (2017) Rdfnet: Rgb-d multi-level residual feature fusion for indoor semantic segmentation. In: 2017 IEEE International Conference on Computer Vision (ICCV), pp 4990\u20134999. https:\/\/doi.org\/10.1109\/ICCV.2017.533","DOI":"10.1109\/ICCV.2017.533"},{"key":"6760_CR6","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2021.108468","author":"H Zhou","year":"2022","unstructured":"Zhou H, Qi L, Huang H, Yang X, Wan Z, Wen X (2022) Canet: co-attention network for rgb-d semantic segmentation. Pattern Recognit. https:\/\/doi.org\/10.1016\/j.patcog.2021.108468","journal-title":"Pattern Recognit"},{"issue":"10","key":"6760_CR7","doi-asserted-by":"publisher","first-page":"2642","DOI":"10.1109\/TPAMI.2019.2923513","volume":"42","author":"D Lin","year":"2020","unstructured":"Lin D, Huang H (2020) Zig-zag network for semantic segmentation of rgb-d images. IEEE Trans Pattern Anal Mach Intell 42(10):2642\u20132655. https:\/\/doi.org\/10.1109\/TPAMI.2019.2923513","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"6760_CR8","doi-asserted-by":"publisher","first-page":"8520","DOI":"10.3390\/s22218520","volume":"22","author":"L Zhu","year":"2022","unstructured":"Zhu L, Kang Z, Zhou M-Y, Yang X, Wang Z, Cao Z, Ye C (2022) Cmanet: cross-modality attention network for indoor-scene semantic segmentation. Sensors (Basel Switz.) 22:8520","journal-title":"Sensors (Basel Switz.)"},{"key":"6760_CR9","doi-asserted-by":"publisher","unstructured":"Seichter D, K\u00f6hler M, Lewandowski B, Wengefeld T, Gross H-M (2021) Efficient rgb-d semantic segmentation for indoor scene analysis. In: 2021 IEEE International Conference on Robotics and Automation (ICRA), pp 13525\u201313531. https:\/\/doi.org\/10.1109\/ICRA48506.2021.9561675","DOI":"10.1109\/ICRA48506.2021.9561675"},{"issue":"3","key":"6760_CR10","doi-asserted-by":"publisher","first-page":"1120","DOI":"10.1109\/TCYB.2018.2885062","volume":"50","author":"D Lin","year":"2020","unstructured":"Lin D, Zhang R, Ji Y, Li P, Huang H (2020) Scn: switchable context network for semantic segmentation of rgb-d images. IEEE Trans Cybern 50(3):1120\u20131131. https:\/\/doi.org\/10.1109\/TCYB.2018.2885062","journal-title":"IEEE Trans Cybern"},{"key":"6760_CR11","doi-asserted-by":"publisher","unstructured":"Cao J, Leng H, Lischinski D, Cohen-Or D, Tu C, Li Y (2021) Shapeconv: shape-aware convolutional layer for indoor rgb-d semantic segmentation. In: 2021 IEEE\/CVF International Conference on Computer Vision (ICCV), pp 7068\u20137077. https:\/\/doi.org\/10.1109\/ICCV48922.2021.00700","DOI":"10.1109\/ICCV48922.2021.00700"},{"key":"6760_CR12","doi-asserted-by":"crossref","unstructured":"Xing Y, Wang J, Zeng G (2020) Malleable 2.5d convolution: learning receptive fields along the depth-axis for rgb-d scene parsing. arXiv:2007.09365","DOI":"10.1007\/978-3-030-58529-7_33"},{"issue":"24","key":"6760_CR13","doi-asserted-by":"publisher","first-page":"24161","DOI":"10.1109\/JSEN.2022.3218601","volume":"22","author":"P Wu","year":"2022","unstructured":"Wu P, Guo R, Tong X, Su S, Zuo Z, Sun B, Wei J (2022) Link-rgbd: cross-guided feature fusion network for rgbd semantic segmentation. IEEE Sensors J 22(24):24161\u201324175. https:\/\/doi.org\/10.1109\/JSEN.2022.3218601","journal-title":"IEEE Sensors J"},{"issue":"19","key":"6760_CR14","doi-asserted-by":"publisher","first-page":"23512","DOI":"10.1109\/JSEN.2023.3304637","volume":"23","author":"Y Zhang","year":"2023","unstructured":"Zhang Y, Xiong C, Liu J, Ye X, Sun G (2023) Spatial information-guided adaptive context-aware network for efficient rgb-d semantic segmentation. IEEE Sensors J 23(19):23512\u201323521. https:\/\/doi.org\/10.1109\/JSEN.2023.3304637","journal-title":"IEEE Sensors J"},{"key":"6760_CR15","doi-asserted-by":"publisher","DOI":"10.1145\/3617827","author":"H Liu","year":"2023","unstructured":"Liu H, Wei Y, Liu F, Wang W, Nie L, Chua T-S (2023) Dynamic multimodal fusion via meta-learning towards micro-video recommendation. ACM Trans Inf Syst. https:\/\/doi.org\/10.1145\/3617827","journal-title":"ACM Trans. Inf. Syst."},{"key":"6760_CR16","doi-asserted-by":"publisher","unstructured":"Gupta S, Girshick R, Arbelaez P, Malik J (2014). Learning rich features from rgb-d images for object detection and segmentation. https:\/\/doi.org\/10.1007\/978-3-319-10584-0_23","DOI":"10.1007\/978-3-319-10584-0_23"},{"key":"6760_CR17","doi-asserted-by":"publisher","unstructured":"Xing Y, Wang J, Chen X, Zeng G (2019) Coupling two-stream rgb-d semantic segmentation network by idempotent mappings. In: 2019 IEEE International Conference on Image Processing (ICIP), pp 1850\u20131854. https:\/\/doi.org\/10.1109\/ICIP.2019.8803146","DOI":"10.1109\/ICIP.2019.8803146"},{"key":"6760_CR18","unstructured":"Jiang J, Zheng L, Luo F, Zhang Z (2018) Rednet: residual encoder-decoder network for indoor rgb-d semantic segmentation. arXiv:1806.01054"},{"key":"6760_CR19","doi-asserted-by":"publisher","first-page":"464","DOI":"10.1016\/j.neucom.2022.04.025","volume":"492","author":"F Zhou","year":"2022","unstructured":"Zhou F, Lai Y-K, Rosin PL, Zhang F, Hu Y (2022) Scale-aware network with modality-awareness for rgb-d indoor semantic segmentation. Neurocomputing 492:464\u2013473","journal-title":"Neurocomputing"},{"key":"6760_CR20","unstructured":"Dosovitskiy A, Beyer L, Kolesnikov A, Weissenborn D, Zhai X, Unterthiner T, Dehghani M, Minderer M, Heigold G, Gelly S, Uszkoreit J, Houlsby N (2020) An image is worth 16x16 words: transformers for image recognition at scale. CoRR arXiv:2010.11929"},{"key":"6760_CR21","doi-asserted-by":"publisher","unstructured":"Liu Z, Lin Y, Cao Y, Hu H, Wei Y, Zhang Z, Lin S, Guo B (2021) Swin transformer: hierarchical vision transformer using shifted windows. In: 2021 IEEE\/CVF International Conference on Computer Vision (ICCV), pp 9992\u201310002. https:\/\/doi.org\/10.1109\/ICCV48922.2021.00986","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"6760_CR22","unstructured":"Xie E, Wang W, Yu Z, Anandkumar A, Alvarez JM, Luo P (2021) Segformer: simple and efficient design for semantic segmentation with transformers. In: Ranzato M, Beygelzimer A, Dauphin Y, Liang PS, Vaughan JW (eds.) Advances in neural information processing systems, vol. 34. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2021\/file\/64f1f27bf1b4ec22924fd0acb550c235-Paper.pdf"},{"issue":"4","key":"6760_CR23","doi-asserted-by":"publisher","first-page":"815","DOI":"10.1109\/TPAMI.2018.2815688","volume":"41","author":"Q Hou","year":"2019","unstructured":"Hou Q, Cheng M-M, Hu X, Borji A, Tu Z, Torr PHS (2019) Deeply supervised salient object detection with short connections. IEEE Trans Pattern Anal Mach Intell 41(4):815\u2013828. https:\/\/doi.org\/10.1109\/TPAMI.2018.2815688","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"6760_CR24","doi-asserted-by":"publisher","unstructured":"Wu Z, Liu L, Zhang Y, Mao M, Lin L, Li G (2022) Multimodal crowd counting with mutual attention transformers. In: 2022 IEEE International Conference on Multimedia and Expo (ICME), pp 1\u20136. https:\/\/doi.org\/10.1109\/ICME52920.2022.9859777","DOI":"10.1109\/ICME52920.2022.9859777"},{"issue":"7","key":"6760_CR25","doi-asserted-by":"publisher","first-page":"1200","DOI":"10.1109\/JAS.2022.105686","volume":"9","author":"J Ma","year":"2022","unstructured":"Ma J, Tang L, Fan F, Huang J, Mei X, Ma Y (2022) Swinfusion: cross-domain long-range learning for general image fusion via swin transformer. IEEE\/CAA J Autom Sin 9(7):1200\u20131217. https:\/\/doi.org\/10.1109\/JAS.2022.105686","journal-title":"IEEE\/CAA J Autom Sin"},{"key":"6760_CR26","doi-asserted-by":"crossref","unstructured":"Yu Z, Zhao C, Wang Z, Qin Y, Su Z, Li X, Zhou F, Zhao G (2020) Searching central difference convolutional networks for face anti-spoofing. In 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp 5294\u20135304","DOI":"10.1109\/CVPR42600.2020.00534"},{"key":"6760_CR27","doi-asserted-by":"crossref","unstructured":"Nathan\u00a0Silberman PK Derek\u00a0Hoiem, Fergus R (2012) Indoor segmentation and support inference from rgbd images. In: ECCV","DOI":"10.1007\/978-3-642-33715-4_54"},{"key":"6760_CR28","doi-asserted-by":"publisher","unstructured":"Cordts M, Omran M, Ramos S, Rehfeld T, Enzweiler M, Benenson R, Franke U, Roth S, Schiele B (2016) The cityscapes dataset for semantic urban scene understanding. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp 3213\u20133223 . https:\/\/doi.org\/10.1109\/CVPR.2016.350","DOI":"10.1109\/CVPR.2016.350"},{"key":"6760_CR29","doi-asserted-by":"publisher","unstructured":"Gupta S. Arbel\u00e1ez P, Malik J (2013) Perceptual organization and recognition of indoor scenes from rgb-d images. In: 2013 IEEE Conference on Computer Vision and Pattern Recognition, pp 564\u2013571. https:\/\/doi.org\/10.1109\/CVPR.2013.79","DOI":"10.1109\/CVPR.2013.79"},{"key":"6760_CR30","doi-asserted-by":"publisher","first-page":"35815","DOI":"10.1007\/s11042-021-11395-w","volume":"81","author":"W Zou","year":"2022","unstructured":"Zou W, Peng Y, Zhang Z, Tian S, Li X (2022) Rgb-d gate-guided edge distillation for indoor semantic segmentation. Multimed Tools App 81:35815\u201335830","journal-title":"Multimed Tools App"},{"issue":"18","key":"6760_CR31","doi-asserted-by":"publisher","first-page":"3943","DOI":"10.3390\/electronics12183943","volume":"12","author":"X Xu","year":"2023","unstructured":"Xu X, Liu J, Liu H (2023) Interactive efficient multi-task network for rgb-d semantic segmentation. Electronics 12(18):3943","journal-title":"Electronics"},{"key":"6760_CR32","doi-asserted-by":"publisher","DOI":"10.1587\/transfun.2024EAP1023","author":"K Zhou","year":"2024","unstructured":"Zhou K, Zhang Z, Tang X, Xu W, Xie J, Tang C (2024) Shape-aware convolution with convolutional kernel attention for rgb-d image semantic segmentation. IEICE Trans Fundam Electron Commun Comput Sci. https:\/\/doi.org\/10.1587\/transfun.2024EAP1023","journal-title":"IEICE Trans Fundam Electron Commun Comput Sci"},{"key":"6760_CR33","doi-asserted-by":"publisher","unstructured":"Cheng Y, Cai R, Li Z, Zhao X, Huang K (2017) Locality-sensitive deconvolution networks with gated fusion for rgb-d indoor semantic segmentation. In: 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp 1475\u20131483. https:\/\/doi.org\/10.1109\/CVPR.2017.161","DOI":"10.1109\/CVPR.2017.161"},{"key":"6760_CR34","doi-asserted-by":"publisher","first-page":"107611","DOI":"10.1016\/j.patcog.2020.107611","volume":"110","author":"M Or\u0161i\u0107","year":"2021","unstructured":"Or\u0161i\u0107 M, \u0160egvi\u0107 S (2021) Efficient semantic segmentation with pyramidal fusion. Pattern Recognit 110:107611. https:\/\/doi.org\/10.1016\/j.patcog.2020.107611","journal-title":"Pattern Recognit"},{"issue":"11","key":"6760_CR35","doi-asserted-by":"publisher","first-page":"3051","DOI":"10.1007\/s11263-021-01515-2","volume":"129","author":"C Yu","year":"2021","unstructured":"Yu C, Gao C, Wang J, Yu G, Shen C, Sang N (2021) Bisenet v2: bilateral network with guided aggregation for real-time semantic segmentation. Int J Comput Vis 129(11):3051\u20133068. https:\/\/doi.org\/10.1007\/s11263-021-01515-2","journal-title":"Int J Comput Vis"},{"key":"6760_CR36","unstructured":"Daquan Z, Hou Q, Chen Y, Feng J, Yan S (2020) Rethinking bottleneck structure for efficient mobile network design. In: European Conference on Computer Vision. https:\/\/api.semanticscholar.org\/CorpusID:220363927"},{"key":"6760_CR37","doi-asserted-by":"publisher","unstructured":"Hung S-W, Lo S-Y, Hang H-M (2019) Incorporating luminance, depth and color information by a fusion-based network for semantic segmentation. In: 2019 IEEE International Conference on Image Processing (ICIP), pp 2374\u20132378. https:\/\/doi.org\/10.1109\/ICIP.2019.8803360","DOI":"10.1109\/ICIP.2019.8803360"},{"issue":"6","key":"6760_CR38","doi-asserted-by":"publisher","first-page":"7293","DOI":"10.1007\/s11227-023-05740-z","volume":"80","author":"Y Song","year":"2023","unstructured":"Song Y, Shang C, Zhao J (2023) Lbcnet: a lightweight bilateral cascaded feature fusion network for real-time semantic segmentation. J Supercomput 80(6):7293\u20137315. https:\/\/doi.org\/10.1007\/s11227-023-05740-z","journal-title":"J Supercomput"},{"key":"6760_CR39","doi-asserted-by":"publisher","unstructured":"Hou J, Dai X, He Z, Dai A, NieBner M (2023) Mask3d: pretraining 2d vision transformers by learning masked 3d priors. In: 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp 13510\u201313519. https:\/\/doi.org\/10.1109\/CVPR52729.2023.01298","DOI":"10.1109\/CVPR52729.2023.01298"},{"key":"6760_CR40","unstructured":"Wang J, Gou C, Wu Q, Feng H, Han J, Ding E, Wang J (2022) Rtformer: efficient design for real-time semantic segmentation with transformer. In: Koyejo S, Mohamed S, Agarwal A, Belgrave D, Cho K, Oh A (eds.) Advances in neural information processing systems, vol 35, pp 7423\u20137436. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2022\/file\/30e10e671c5e43edb67eb257abb6c3ea-Paper-Conference.pdf"},{"key":"6760_CR41","doi-asserted-by":"crossref","unstructured":"Zhang W, Huang Z, Luo G, Chen T, Wang X, Liu W, Yu G, Shen C (2022) Topformer: token pyramid transformer for mobile semantic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 12083\u201312093","DOI":"10.1109\/CVPR52688.2022.01177"},{"key":"6760_CR42","unstructured":"Wan Q, Huang Z, Lu J, Yu G, Zhang L (2023) Seaformer: squeeze-enhanced axial transformer for mobile semantic segmentation. arXiv:2301.13156"},{"key":"6760_CR43","doi-asserted-by":"crossref","unstructured":"Dong B, Wang P, Wang F (2023) Head-free lightweight semantic segmentation with linear transformer. In: AAAI Conference on Artificial Intelligence. https:\/\/api.semanticscholar.org\/CorpusID:255595564","DOI":"10.1609\/aaai.v37i1.25126"},{"key":"6760_CR44","doi-asserted-by":"crossref","unstructured":"Xu Z, Wu D, Yu C, Chu X, Sang N, Gao C (2024) Sctnet: single-branch cnn with transformer semantic information for real-time segmentation. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 38, pp 6378\u20136386","DOI":"10.1609\/aaai.v38i6.28457"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-024-06760-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-024-06760-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-024-06760-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,4]],"date-time":"2024-12-04T12:09:29Z","timestamp":1733314169000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-024-06760-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,4]]},"references-count":44,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2025,1]]}},"alternative-id":["6760"],"URL":"https:\/\/doi.org\/10.1007\/s11227-024-06760-z","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"type":"print","value":"0920-8542"},{"type":"electronic","value":"1573-0484"}],"subject":[],"published":{"date-parts":[[2024,12,4]]},"assertion":[{"value":"20 November 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 December 2024","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"242"}}