{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T20:18:35Z","timestamp":1783196315712,"version":"3.54.6"},"reference-count":42,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100018537","name":"National Science and Technology Major Project","doi-asserted-by":"publisher","award":["2025ZD1608102"],"award-info":[{"award-number":["2025ZD1608102"]}],"id":[{"id":"10.13039\/501100018537","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Journal of Visual Communication and Image Representation"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1016\/j.jvcir.2026.104822","type":"journal-article","created":{"date-parts":[[2026,4,24]],"date-time":"2026-04-24T15:50:34Z","timestamp":1777045834000},"page":"104822","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["ASNet: An adaptive scene-aware network for RGB-thermal urban scene semantic segmentation"],"prefix":"10.1016","volume":"118","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-0275-2406","authenticated-orcid":false,"given":"Yixin","family":"Guo","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9637-5170","authenticated-orcid":false,"given":"Zhenxue","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3283-5347","authenticated-orcid":false,"given":"Xuewen","family":"Rong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chengyun","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lili","family":"Song","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yidi","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.jvcir.2026.104822_b1","doi-asserted-by":"crossref","first-page":"22761","DOI":"10.1109\/TITS.2025.3615568","article-title":"Sl-seg: A cnn-transformer fusion network for road surface and lane segmentation in complex scenarios","volume":"26","author":"Meng","year":"2025","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"10.1016\/j.jvcir.2026.104822_b2","doi-asserted-by":"crossref","DOI":"10.1016\/j.jvcir.2024.104188","article-title":"Bsnet: A bilateral real-time semantic segmentation network based on multi-scale receptive fields","volume":"102","author":"Jin","year":"2024","journal-title":"J. Vis. Commun. Image Represent."},{"key":"10.1016\/j.jvcir.2026.104822_b3","doi-asserted-by":"crossref","first-page":"29503","DOI":"10.1109\/ACCESS.2024.3369039","article-title":"Optimizing road traffic surveillance: A robust hyper-heuristic approach for vehicle segmentation","volume":"12","author":"Rodr\u00edguez-Esparza","year":"2024","journal-title":"IEEE Access"},{"key":"10.1016\/j.jvcir.2026.104822_b4","doi-asserted-by":"crossref","first-page":"2173","DOI":"10.1007\/s13369-022-07092-x","article-title":"Semantic segmentation based crowd tracking and anomaly detection via neuro-fuzzy classifier in smart surveillance system","volume":"48","author":"Abdullah","year":"2023","journal-title":"Arab. J. Sci. Eng."},{"key":"10.1016\/j.jvcir.2026.104822_b5","doi-asserted-by":"crossref","first-page":"10076","DOI":"10.1109\/TPAMI.2024.3435571","article-title":"Medical image segmentation review: The success of u-net","volume":"46","author":"Azad","year":"2024","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.jvcir.2026.104822_b6","doi-asserted-by":"crossref","DOI":"10.1016\/j.measurement.2021.110176","article-title":"Robust semantic segmentation based on rgb-thermal in variable lighting scenes","volume":"186","author":"Guo","year":"2021","journal-title":"Measurement"},{"key":"10.1016\/j.jvcir.2026.104822_b7","doi-asserted-by":"crossref","unstructured":"H. Wang, Y. Sun, R. Fan, M. Liu, S2p2: Self-supervised goal-directed path planning using rgb-d data for robotic wheelchairs, in: 2021 IEEE International Conference on Robotics and Automation, 2021, pp. 11422\u201311428.","DOI":"10.1109\/ICRA48506.2021.9561314"},{"key":"10.1016\/j.jvcir.2026.104822_b8","doi-asserted-by":"crossref","first-page":"4687","DOI":"10.1109\/TIV.2024.3376534","article-title":"Segmentation of road negative obstacles based on dual semantic-feature complementary fusion for autonomous driving","volume":"9","author":"Feng","year":"2024","journal-title":"IEEE Trans. Intell. Veh."},{"key":"10.1016\/j.jvcir.2026.104822_b9","doi-asserted-by":"crossref","unstructured":"Z. Huang, X. Wang, L. Huang, C. Huang, Y. Wei, W. Liu, Ccnet: Criss-cross attention for semantic segmentation, in: ICCV, 2019, pp. 603\u2013612.","DOI":"10.1109\/ICCV.2019.00069"},{"key":"10.1016\/j.jvcir.2026.104822_b10","doi-asserted-by":"crossref","unstructured":"C. Yu, J. Wang, C. Peng, C. Gao, G. Yu, N. Sang, Learning a discriminative feature network for semantic segmentation, in: CVPR, 2018, pp. 1857\u20131866.","DOI":"10.1109\/CVPR.2018.00199"},{"key":"10.1016\/j.jvcir.2026.104822_b11","doi-asserted-by":"crossref","first-page":"7012","DOI":"10.1109\/TIP.2020.3028289","article-title":"Dpanet: Depth potentiality-aware gated attention network for rgb-d salient object detection","volume":"30","author":"Chen","year":"2020","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.jvcir.2026.104822_b12","doi-asserted-by":"crossref","DOI":"10.1016\/j.jvcir.2025.104523","article-title":"Lightweight three-stream encoder-decoder network for multi-modal salient object detection","volume":"111","author":"Lu","year":"2025","journal-title":"J. Vis. Commun. Image Represent."},{"key":"10.1016\/j.jvcir.2026.104822_b13","doi-asserted-by":"crossref","DOI":"10.1016\/j.jvcir.2025.104622","article-title":"Multiple cross-modal complementation network for lightweight rgb-d salient object detection","volume":"113","author":"Zhang","year":"2025","journal-title":"J. Vis. Commun. Image Represent."},{"key":"10.1016\/j.jvcir.2026.104822_b14","doi-asserted-by":"crossref","unstructured":"Q. Ha, K. Watanabe, T. Karasawa, Y. Ushiku, T. Harada, Mfnet: Towards real-time semantic segmentation for autonomous vehicles with multi-spectral scenes, in: 2017 IEEE\/RSJ International Conference on Intelligent Robots and Systems, 2017, pp. 5108\u20135115.","DOI":"10.1109\/IROS.2017.8206396"},{"key":"10.1016\/j.jvcir.2026.104822_b15","doi-asserted-by":"crossref","first-page":"1075","DOI":"10.1007\/s11760-025-04672-w","article-title":"Hierarchical multi-modal feature fusion for rgbt tracking","volume":"19","author":"Li","year":"2025","journal-title":"Signal, Image Video Process."},{"key":"10.1016\/j.jvcir.2026.104822_b16","doi-asserted-by":"crossref","unstructured":"S.S. Shivakumar, N. Rodrigues, A. Zhou, I.D. Miller, V. Kumar, C.J. Taylor, Pst900: Rgb-thermal calibration, dataset and segmentation network, in: 2020 IEEE International Conference on Robotics and Automation, 2020, pp. 9441\u20139447.","DOI":"10.1109\/ICRA40945.2020.9196831"},{"key":"10.1016\/j.jvcir.2026.104822_b17","doi-asserted-by":"crossref","first-page":"2576","DOI":"10.1109\/LRA.2019.2904733","article-title":"Rtfnet: Rgb-thermal fusion network for semantic segmentation of urban scenes","volume":"4","author":"Sun","year":"2019","journal-title":"IEEE Robot. Autom. Lett."},{"key":"10.1016\/j.jvcir.2026.104822_b18","doi-asserted-by":"crossref","unstructured":"F. Deng, H. Feng, M. Liang, H. Wang, Y. Yang, Y. Gao, J. Chen, J. Hu, X. Guo, T.L. Lam, Feanet: Feature-enhanced attention network for rgb-thermal real-time semantic segmentation, in: 2021 IEEE\/RSJ International Conference on Intelligent Robots and Systems, 2021, pp. 4467\u20134473.","DOI":"10.1109\/IROS51168.2021.9636084"},{"key":"10.1016\/j.jvcir.2026.104822_b19","doi-asserted-by":"crossref","first-page":"2526","DOI":"10.1109\/TMM.2021.3086618","article-title":"Mffenet: Multiscale feature fusion and enhancement network for rgb-thermal urban road scene parsing","volume":"24","author":"Zhou","year":"2021","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.jvcir.2026.104822_b20","doi-asserted-by":"crossref","first-page":"7790","DOI":"10.1109\/TIP.2021.3109518","article-title":"Gmnet: Graded-feature multilabel-learning network for rgb-thermal urban scene semantic segmentation","volume":"30","author":"Zhou","year":"2021","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.jvcir.2026.104822_b21","doi-asserted-by":"crossref","first-page":"1223","DOI":"10.1109\/TCSVT.2022.3208833","article-title":"Rgb-t semantic segmentation with location, activation, and sharpening","volume":"33","author":"Li","year":"2023","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.jvcir.2026.104822_b22","doi-asserted-by":"crossref","unstructured":"J. Long, E. Shelhamer, T. Darrell, Fully convolutional networks for semantic segmentation, in: CVPR, 2015, pp. 3431\u20133440.","DOI":"10.1109\/CVPR.2015.7298965"},{"key":"10.1016\/j.jvcir.2026.104822_b23","doi-asserted-by":"crossref","unstructured":"L.-C. Chen, Y. Zhu, G. Papandreou, F. Schroff, H. Adam, Encoder-decoder with atrous separable convolution for semantic image segmentation, in: ECCV, 2018, pp. 801\u2013818.","DOI":"10.1007\/978-3-030-01234-2_49"},{"key":"10.1016\/j.jvcir.2026.104822_b24","doi-asserted-by":"crossref","unstructured":"H. Zhao, J. Shi, X. Qi, X. Wang, J. Jia, Pyramid scene parsing network, in: CVPR, 2017, pp. 2881\u20132890.","DOI":"10.1109\/CVPR.2017.660"},{"key":"10.1016\/j.jvcir.2026.104822_b25","doi-asserted-by":"crossref","unstructured":"C. Yu, J. Wang, C. Peng, C. Gao, G. Yu, N. Sang, Bisnet: Bilateral segmentation network for real-time semantic segmentation, in: ECCV, 2018, pp. 325\u2013341.","DOI":"10.1007\/978-3-030-01261-8_20"},{"key":"10.1016\/j.jvcir.2026.104822_b26","doi-asserted-by":"crossref","unstructured":"J. Fu, J. Liu, H. Tian, Y. Li, Y. Bao, Z. Fang, H. Lu, Dual attention network for scene segmentation, in: CVPR, 2019, pp. 3146\u20133154.","DOI":"10.1109\/CVPR.2019.00326"},{"key":"10.1016\/j.jvcir.2026.104822_b27","doi-asserted-by":"crossref","DOI":"10.1016\/j.jvcir.2023.104028","article-title":"Bndcnet: Bilateral nonlocal decoupled convergence network for semantic segmentation","volume":"98","author":"Ye","year":"2024","journal-title":"J. Vis. Commun. Image Represent."},{"key":"10.1016\/j.jvcir.2026.104822_b28","first-page":"12077","article-title":"Segformer: Simple and efficient design for semantic segmentation with transformers","volume":"vol. 34","author":"Xie","year":"2021"},{"key":"10.1016\/j.jvcir.2026.104822_b29","doi-asserted-by":"crossref","unstructured":"A. Kirillov, E. Mintun, N. Ravi, H. Mao, C. Rolland, L. Gustafson, T. Xiao, S. Whitehead, A.C. Berg, W.-Y. Lo, et al., Segment anything, in: ICCV, 2023, pp. 4015\u20134026.","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"10.1016\/j.jvcir.2026.104822_b30","doi-asserted-by":"crossref","unstructured":"H. Kweon, K.-J. Yoon, From sam to cams: Exploring segment anything model for weakly supervised semantic segmentation, in: CVPR, 2024, pp. 19499\u201319509.","DOI":"10.1109\/CVPR52733.2024.01844"},{"key":"10.1016\/j.jvcir.2026.104822_b31","unstructured":"J.-J. Wu, A.C.-H. Chang, C.-Y. Chuang, C.-P. Chen, Y.-L. Liu, M.-H. Chen, H.-N. Hu, Y.-Y. Chuang, Y.-Y. Lin, Image-text co-decomposition for text-supervised semantic segmentation, in: CVPR, 2024, pp. 26794\u201326803."},{"key":"10.1016\/j.jvcir.2026.104822_b32","doi-asserted-by":"crossref","unstructured":"X. Hu, L. Jiang, B. Schiele, Training vision transformers for semi-supervised semantic segmentation, in: CVPR, 2024, pp. 4007\u20134017.","DOI":"10.1109\/CVPR52733.2024.00384"},{"key":"10.1016\/j.jvcir.2026.104822_b33","doi-asserted-by":"crossref","first-page":"1000","DOI":"10.1109\/TASE.2020.2993143","article-title":"Fuseseg: Semantic segmentation of urban scenes based on rgb and thermal data fusion","volume":"18","author":"Sun","year":"2021","journal-title":"IEEE Trans. Autom. Sci. Eng."},{"key":"10.1016\/j.jvcir.2026.104822_b34","doi-asserted-by":"crossref","first-page":"5817","DOI":"10.1007\/s10489-021-02687-7","article-title":"Mmnet: Multi-modal multi-stage network for rgbt image semantic segmentation","volume":"52","author":"Lan","year":"2022","journal-title":"Appl. Intell."},{"key":"10.1016\/j.jvcir.2026.104822_b35","doi-asserted-by":"crossref","unstructured":"Y. Guo, Z. Chen, X. Rong, C. Liu, Y. Li, Rfnet: A redundancy-reduced and frequency-aware feature fusion network for rgb-t semantic segmentation, in: 2024 IEEE Smart World Congress, 2024, pp. 47\u201351.","DOI":"10.1109\/SWC62898.2024.00037"},{"key":"10.1016\/j.jvcir.2026.104822_b36","doi-asserted-by":"crossref","unstructured":"W. Zhou, S. Dong, C. Xu, Y. Qian, Edge-aware guidance fusion network for rgb-thermal scene parsing, in: AAAI, 36, 2022, pp. 3571\u20133579.","DOI":"10.1609\/aaai.v36i3.20269"},{"key":"10.1016\/j.jvcir.2026.104822_b37","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2024.127913","article-title":"Residual spatial fusion network for rgb-thermal semantic segmentation","volume":"595","author":"Li","year":"2024","journal-title":"Neurocomputing"},{"key":"10.1016\/j.jvcir.2026.104822_b38","doi-asserted-by":"crossref","unstructured":"K. He, X. Zhang, S. Ren, J. Sun, Deep residual learning for image recognition, in: CVPR, 2016, pp. 770\u2013778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"10.1016\/j.jvcir.2026.104822_b39","doi-asserted-by":"crossref","first-page":"3522","DOI":"10.1109\/TITS.2020.3037727","article-title":"Frnet: Factorized and regular blocks network for semantic segmentation in road scene","volume":"23","author":"Lu","year":"2020","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"10.1016\/j.jvcir.2026.104822_b40","first-page":"23","article-title":"A threshold selection method from gray-level histograms","volume":"11","author":"Otsu","year":"1975","journal-title":"Automatica"},{"key":"10.1016\/j.jvcir.2026.104822_b41","doi-asserted-by":"crossref","unstructured":"K. Han, Y. Wang, Q. Tian, J. Guo, C. Xu, C. Xu, Ghostnet: More features from cheap operations, in: CVPR, 2020, pp. 1580\u20131589.","DOI":"10.1109\/CVPR42600.2020.00165"},{"key":"10.1016\/j.jvcir.2026.104822_b42","doi-asserted-by":"crossref","unstructured":"C. Hazirbas, L. Ma, C. Domokos, D. Cremers, Fusenet: Incorporating depth into semantic segmentation via fusion-based cnn architecture, in: Asian Conference on Computer Vision, 2017, pp. 213\u2013228.","DOI":"10.1007\/978-3-319-54181-5_14"}],"container-title":["Journal of Visual Communication and Image Representation"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1047320326001173?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1047320326001173?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T19:54:51Z","timestamp":1783194891000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1047320326001173"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6]]},"references-count":42,"alternative-id":["S1047320326001173"],"URL":"https:\/\/doi.org\/10.1016\/j.jvcir.2026.104822","relation":{},"ISSN":["1047-3203"],"issn-type":[{"value":"1047-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,6]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"ASNet: An adaptive scene-aware network for RGB-thermal urban scene semantic segmentation","name":"articletitle","label":"Article Title"},{"value":"Journal of Visual Communication and Image Representation","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.jvcir.2026.104822","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Inc. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"104822"}}