{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T18:47:58Z","timestamp":1777488478521,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":19,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,3,25]],"date-time":"2022-03-25T00:00:00Z","timestamp":1648166400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"NSFC","award":["61203172"],"award-info":[{"award-number":["61203172"]}]},{"name":"the Sichuan Science and Technology Programs","award":["2019YFH0187"],"award-info":[{"award-number":["2019YFH0187"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,3,25]]},"DOI":"10.1145\/3532342.3532352","type":"proceedings-article","created":{"date-parts":[[2022,6,29]],"date-time":"2022-06-29T22:12:48Z","timestamp":1656540768000},"page":"68-73","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":11,"title":["Hierarchical Vision Transformer with Channel Attention for RGB-D Image Segmentation"],"prefix":"10.1145","author":[{"given":"Yali","family":"Yang","sequence":"first","affiliation":[{"name":"School of Software Engineering, Chengdu University of Information Technology, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuanping","family":"Xu","sequence":"additional","affiliation":[{"name":"School of Software Engineering, Chengdu University of Information Technology, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chaolong","family":"Zhang","sequence":"additional","affiliation":[{"name":"School of Computing and Engineering, University of Huddersfield, UK"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhijie","family":"Xu","sequence":"additional","affiliation":[{"name":"School of Computing and Engineering, University of Huddersfield, UK"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jian","family":"Huang","sequence":"additional","affiliation":[{"name":"School of Software Engineering, Chengdu University of Information Technology, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2022,6,29]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Vaswani A Shazeer N Parmar N Attention is all you need[C]\/\/Advances in neural information processing systems. 2017: 5998-6008."},{"key":"e_1_3_2_1_2_1","volume-title":"An image is worth 16x16 words: Transformers for image recognition at scale[J]. arXiv preprint arXiv:2010.11929","author":"Dosovitskiy A","year":"2020","unstructured":"Dosovitskiy A, Beyer L, Kolesnikov A, An image is worth 16x16 words: Transformers for image recognition at scale[J]. arXiv preprint arXiv:2010.11929, 2020."},{"key":"e_1_3_2_1_3_1","volume-title":"Gu S","author":"Peng Z","year":"2021","unstructured":"Peng Z, Huang W, Gu S, Conformer: Local Features Coupling Global Representations for Visual Recognition[J]. arXiv preprint arXiv:2105.03889, 2021."},{"key":"e_1_3_2_1_4_1","volume-title":"Laptev I","author":"Strudel R","year":"2021","unstructured":"Strudel R, Garcia R, Laptev I, Segmenter: Transformer for Semantic Segmentation[J]. arXiv preprint arXiv:2105.05633, 2021."},{"key":"e_1_3_2_1_5_1","volume-title":"Swin transformer: Hierarchical vision transformer using shifted windows[J]. arXiv preprint arXiv:2103.14030","author":"Liu Z","year":"2021","unstructured":"Liu Z, Lin Y, Cao Y, Swin transformer: Hierarchical vision transformer using shifted windows[J]. arXiv preprint arXiv:2103.14030, 2021."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"crossref","unstructured":"Hu J Shen L Sun G. Squeeze-and-excitation networks[C]\/\/Proceedings of the IEEE conference on computer vision and pattern recognition. 2018: 7132-7141.","DOI":"10.1109\/CVPR.2018.00745"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"crossref","unstructured":"Long J Shelhamer E Darrell T. Fully convolutional networks for semantic segmentation[C]\/\/Proceedings of the IEEE conference on computer vision and pattern recognition. 2015: 3431-3440.","DOI":"10.1109\/CVPR.2015.7298965"},{"key":"e_1_3_2_1_8_1","volume-title":"Brox T. U-net: Convolutional networks for biomedical image segmentation[C]\/\/International Conference on Medical image computing and computer-assisted intervention","author":"Ronneberger O","year":"2015","unstructured":"Ronneberger O, Fischer P, Brox T. U-net: Convolutional networks for biomedical image segmentation[C]\/\/International Conference on Medical image computing and computer-assisted intervention. Springer, Cham, 2015: 234-241."},{"key":"e_1_3_2_1_9_1","volume-title":"Indoor segmentation and support inference from rgbd images[C]\/\/European conference on computer vision","author":"Silberman N","year":"2012","unstructured":"Silberman N, Hoiem D, Kohli P, Indoor segmentation and support inference from rgbd images[C]\/\/European conference on computer vision. Springer, Berlin, Heidelberg, 2012: 746-760."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"crossref","unstructured":"Song S Lichtenberg S P Xiao J. Sun rgb-d: A rgb-d scene understanding benchmark suite[C]\/\/Proceedings of the IEEE conference on computer vision and pattern recognition. 2015: 567-576.","DOI":"10.1109\/CVPR.2015.7298655"},{"key":"e_1_3_2_1_11_1","volume-title":"A category-level 3d object dataset: Putting the kinect to work[M]\/\/Consumer depth cameras for computer vision","author":"Janoch A","year":"2013","unstructured":"Janoch A, Karayev S, Jia Y, A category-level 3d object dataset: Putting the kinect to work[M]\/\/Consumer depth cameras for computer vision. Springer, London, 2013: 141-165."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"crossref","unstructured":"Xiao J Owens A Torralba A. Sun3d: A database of big spaces reconstructed using sfm and object labels[C]\/\/Proceedings of the IEEE international conference on computer vision. 2013: 1625-1632.","DOI":"10.1109\/ICCV.2013.458"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"crossref","unstructured":"Long J Shelhamer E Darrell T. Fully convolutional networks for semantic segmentation[C]\/\/Proceedings of the IEEE conference on computer vision and pattern recognition. 2015: 3431-3440.","DOI":"10.1109\/CVPR.2015.7298965"},{"key":"e_1_3_2_1_14_1","volume-title":"Exploring context with deep structured models for semantic segmentation[J]","author":"Lin G","year":"2017","unstructured":"Lin G, Shen C, Van Den Hengel A, Exploring context with deep structured models for semantic segmentation[J]. IEEE transactions on pattern analysis and machine intelligence, 2017, 40(6): 1352-1366."},{"key":"e_1_3_2_1_15_1","volume-title":"Shen C","author":"Lin G","year":"2017","unstructured":"Lin G, Milan A, Shen C, Refinenet: Multi-path refinement networks for high-resolution semantic segmentation[C]\/\/Proceedings of the IEEE conference on computer vision and pattern recognition. 2017: 1925-1934."},{"key":"e_1_3_2_1_16_1","unstructured":"Park S J Hong K S Lee S. Rdfnet: Rgb-d multi-level residual feature fusion for indoor semantic segmentation[C]\/\/Proceedings of the IEEE international conference on computer vision. 2017: 4980-4989."},{"key":"e_1_3_2_1_17_1","volume-title":"Liang X","author":"Li Z","year":"2016","unstructured":"Li Z, Gan Y, Liang X, Lstm-cf: Unifying context modeling and fusion with lstms for rgb-d scene labeling[C]\/\/European conference on computer vision. Springer, Cham, 2016: 541-557."},{"key":"e_1_3_2_1_18_1","volume-title":"Domokos C","author":"Hazirbas C","year":"2016","unstructured":"Hazirbas C, Ma L, Domokos C, Fusenet: Incorporating depth into semantic segmentation via fusion-based cnn architecture[C]\/\/Asian conference on computer vision. Springer, Cham, 2016: 213-228."},{"key":"e_1_3_2_1_19_1","volume-title":"Bayesian segnet: Model uncertainty in deep convolutional encoder-decoder architectures for scene understanding[J]. arXiv preprint arXiv:1511.02680","author":"Kendall A","year":"2015","unstructured":"Kendall A, Badrinarayanan V, Cipolla R. Bayesian segnet: Model uncertainty in deep convolutional encoder-decoder architectures for scene understanding[J]. arXiv preprint arXiv:1511.02680, 2015."}],"event":{"name":"SSPS 2022: 2022 4th International Symposium on Signal Processing Systems","location":"Xi'an China","acronym":"SSPS 2022"},"container-title":["Proceedings of the 4th International Symposium on Signal Processing Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3532342.3532352","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3532342.3532352","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T19:30:09Z","timestamp":1750188609000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3532342.3532352"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,3,25]]},"references-count":19,"alternative-id":["10.1145\/3532342.3532352","10.1145\/3532342"],"URL":"https:\/\/doi.org\/10.1145\/3532342.3532352","relation":{},"subject":[],"published":{"date-parts":[[2022,3,25]]},"assertion":[{"value":"2022-06-29","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}