{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,27]],"date-time":"2026-04-27T11:54:54Z","timestamp":1777290894897,"version":"3.51.4"},"reference-count":63,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Image and Vision Computing"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1016\/j.imavis.2026.105977","type":"journal-article","created":{"date-parts":[[2026,4,6]],"date-time":"2026-04-06T01:11:16Z","timestamp":1775437876000},"page":"105977","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Distilling auxiliary RGB\u2013T features for unsupervised semantic segmentation"],"prefix":"10.1016","volume":"170","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-7007-6443","authenticated-orcid":false,"given":"S. Meena","family":"Padnekar","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kaushik","family":"Mitra","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sukhendu","family":"Das","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.imavis.2026.105977_b1","series-title":"2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops","first-page":"700","article-title":"A comparative study of real-time semantic segmentation for autonomous driving","author":"Siam","year":"2018"},{"issue":"10","key":"10.1016\/j.imavis.2026.105977_b2","doi-asserted-by":"crossref","DOI":"10.1016\/j.jksuci.2024.102226","article-title":"Real-time semantic segmentation for autonomous driving: A review of cnns, transformers, and beyond","volume":"36","author":"Elhassan","year":"2024","journal-title":"J. King Saud Univ. - Comput. Inf. Sci."},{"key":"10.1016\/j.imavis.2026.105977_b3","doi-asserted-by":"crossref","DOI":"10.1016\/j.media.2023.102918","article-title":"Segment anything model for medical image analysis: An experimental study","volume":"89","author":"Mazurowski","year":"2023","journal-title":"Med. Image Anal."},{"key":"10.1016\/j.imavis.2026.105977_b4","doi-asserted-by":"crossref","DOI":"10.1016\/j.imavis.2024.105371","article-title":"Dual multi scale networks for medical image segmentation using contrastive learning","volume":"154","author":"Dhamale","year":"2025","journal-title":"Image Vis. Comput."},{"key":"10.1016\/j.imavis.2026.105977_b5","first-page":"12077","article-title":"Segformer: Simple and efficient design for semantic segmentation with transformers","volume":"vol. 34","author":"Xie","year":"2021"},{"key":"10.1016\/j.imavis.2026.105977_b6","first-page":"17864","article-title":"Per-pixel classification is not all you need for semantic segmentation","volume":"vol. 34","author":"Cheng","year":"2021"},{"key":"10.1016\/j.imavis.2026.105977_b7","series-title":"2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"6877","article-title":"Rethinking semantic segmentation from a sequence-to-sequence perspective with transformers","author":"Zheng","year":"2021"},{"key":"10.1016\/j.imavis.2026.105977_b8","series-title":"Computer Vision \u2013 ECCV 2014","first-page":"740","article-title":"Microsoft coco: Common objects in context","author":"Lin","year":"2014"},{"key":"10.1016\/j.imavis.2026.105977_b9","unstructured":"M. Hamilton, Z. Zhang, B. Hariharan, N. Snavely, W.T. Freeman, Unsupervised semantic segmentation by distilling feature correspondences, in: International Conference on Learning Representations, 2022."},{"key":"10.1016\/j.imavis.2026.105977_b10","series-title":"2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"19540","article-title":"Leveraging hidden positives for unsupervised semantic segmentation","author":"Seong","year":"2023"},{"key":"10.1016\/j.imavis.2026.105977_b11","series-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems","article-title":"Smooseg: smoothness prior for unsupervised semantic segmentation","author":"Lan","year":"2024"},{"key":"10.1016\/j.imavis.2026.105977_b12","series-title":"2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"3523","article-title":"EAGLE: Eigen Aggregation Learning for Object-Centric Unsupervised Semantic Segmentation","author":"Kim","year":"2024"},{"key":"10.1016\/j.imavis.2026.105977_b13","series-title":"2021 IEEE\/CVF International Conference on Computer Vision","first-page":"9630","article-title":"Emerging properties in self-supervised vision transformers","author":"Caron","year":"2021"},{"key":"10.1016\/j.imavis.2026.105977_b14","series-title":"2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"7162","article-title":"ACSeg: Adaptive conceptualization for unsupervised semantic segmentation","author":"Li","year":"2023"},{"key":"10.1016\/j.imavis.2026.105977_b15","unstructured":"A. Zadaianchuk, M. Kleindessner, Y. Zhu, F. Locatello, T. Brox, Unsupervised semantic segmentation with self-supervised object-centric representations, in: The Eleventh International Conference on Learning Representations, 2023."},{"key":"10.1016\/j.imavis.2026.105977_b16","series-title":"2021 IEEE\/CVF International Conference on Computer Vision","first-page":"10032","article-title":"Unsupervised semantic segmentation by contrasting object mask proposals","author":"Gansbeke","year":"2021"},{"key":"10.1016\/j.imavis.2026.105977_b17","series-title":"Discovering object masks with transformers for unsupervised semantic segmentation","author":"Gansbeke","year":"2022"},{"key":"10.1016\/j.imavis.2026.105977_b18","unstructured":"A. Dosovitskiy, L. Beyer, A. Kolesnikov, D. Weissenborn, X. Zhai, T. Unterthiner, M. Dehghani, M. Minderer, G. Heigold, S. Gelly, J. Uszkoreit, N. Houlsby, An image is worth 16x16 words: Transformers for image recognition at scale, in: International Conference on Learning Representations, 2021."},{"key":"10.1016\/j.imavis.2026.105977_b19","series-title":"2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"1209","article-title":"COCO-Stuff: Thing and Stuff Classes in Context","author":"Caesar","year":"2018"},{"key":"10.1016\/j.imavis.2026.105977_b20","series-title":"2016 IEEE Conference on Computer Vision and Pattern Recognition","first-page":"3213","article-title":"The cityscapes dataset for semantic urban scene understanding","author":"Cordts","year":"2016"},{"key":"10.1016\/j.imavis.2026.105977_b21","series-title":"2017 IEEE\/RSJ International Conference on Intelligent Robots and Systems","first-page":"5108","article-title":"Mfnet: Towards real-time semantic segmentation for autonomous vehicles with multi-spectral scenes","author":"Ha","year":"2017"},{"key":"10.1016\/j.imavis.2026.105977_b22","doi-asserted-by":"crossref","DOI":"10.1016\/j.cviu.2025.104421","article-title":"Ertfnet: Enhanced rgb-t fusion network for semantic segmentation by integrating thermal edge features","volume":"259","author":"Yin","year":"2025","journal-title":"Comput. Vis. Image Underst."},{"key":"10.1016\/j.imavis.2026.105977_b23","series-title":"2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"2633","article-title":"Abmdrnet: Adaptive-weighted bi-directional modality difference reduction network for rgb-t semantic segmentation","author":"Zhang","year":"2021"},{"issue":"3","key":"10.1016\/j.imavis.2026.105977_b24","doi-asserted-by":"crossref","first-page":"2576","DOI":"10.1109\/LRA.2019.2904733","article-title":"Rtfnet: Rgb-thermal fusion network for semantic segmentation of urban scenes","volume":"4","author":"Sun","year":"2019","journal-title":"IEEE Robot. Autom. Lett."},{"issue":"12","key":"10.1016\/j.imavis.2026.105977_b25","doi-asserted-by":"crossref","first-page":"14679","DOI":"10.1109\/TITS.2023.3300537","article-title":"Cmx: Cross-modal fusion for rgb-x semantic segmentation with transformers","volume":"24","author":"Zhang","year":"2023","journal-title":"Trans. Intell. Transp. Sys."},{"key":"10.1016\/j.imavis.2026.105977_b26","series-title":"Csfnet: A cosine similarity fusion network for real-time rgb-x semantic segmentation of driving scenes","author":"Qashqai","year":"2024"},{"key":"10.1016\/j.imavis.2026.105977_b27","series-title":"2024 IEEE International Conference on Robotics and Automation","first-page":"11110","article-title":"Complementary random masking for rgb-thermal semantic segmentation","author":"Shin","year":"2024"},{"key":"10.1016\/j.imavis.2026.105977_b28","series-title":"Computer Vision \u2013 ECCV 2020","first-page":"646","article-title":"A single stream network for robust and real-time rgb-d salient object detection","author":"Zhao","year":"2020"},{"issue":"9","key":"10.1016\/j.imavis.2026.105977_b29","first-page":"5761","article-title":"Uncertainty inspired rgb-d saliency detection","volume":"44","author":"Zhang","year":"2022","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.imavis.2026.105977_b30","series-title":"Computer Vision \u2013 ACCV 2016","first-page":"213","article-title":"Fusenet: Incorporating depth into semantic segmentation via fusion-based cnn architecture","author":"Hazirbas","year":"2017"},{"key":"10.1016\/j.imavis.2026.105977_b31","series-title":"2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"175","article-title":"Deep depth completion of a single rgb-d image","author":"Zhang","year":"2018"},{"issue":"11","key":"10.1016\/j.imavis.2026.105977_b32","doi-asserted-by":"crossref","first-page":"12878","DOI":"10.1109\/TPAMI.2022.3200245","article-title":"Transfuser: Imitation with transformer-based sensor fusion for autonomous driving","volume":"45","author":"Chitta","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.imavis.2026.105977_b33","series-title":"2017 IEEE International Conference on Computer Vision","first-page":"4990","article-title":"Rdfnet: Rgb-d multi-level residual feature fusion for indoor semantic segmentation","author":"Lee","year":"2017"},{"key":"10.1016\/j.imavis.2026.105977_b34","series-title":"2019 IEEE International Conference on Image Processing","first-page":"1440","article-title":"Acnet: Attention based network to exploit complementary features for rgbd semantic segmentation","author":"Hu","year":"2019"},{"key":"10.1016\/j.imavis.2026.105977_b35","series-title":"2021 IEEE\/CVF International Conference on Computer Vision","first-page":"4702","author":"Liu","year":"2021"},{"key":"10.1016\/j.imavis.2026.105977_b36","series-title":"Computer Vision \u2013 ECCV 2018","first-page":"3","article-title":"Cbam: Convolutional block attention module","author":"Woo","year":"2018"},{"key":"10.1016\/j.imavis.2026.105977_b37","series-title":"2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"11531","article-title":"ECA-Net: Efficient Channel Attention for Deep Convolutional Neural Networks","author":"Wang","year":"2020"},{"key":"10.1016\/j.imavis.2026.105977_b38","first-page":"4835","article-title":"Deep multimodal fusion by channel exchanging","volume":"vol. 33","author":"Wang","year":"2020"},{"key":"10.1016\/j.imavis.2026.105977_b39","series-title":"2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"12176","article-title":"Multimodal Token Fusion for Vision Transformers","author":"Wang","year":"2022"},{"key":"10.1016\/j.imavis.2026.105977_b40","series-title":"2024 IEEE\/CVF Winter Conference on Applications of Computer Vision","first-page":"1009","article-title":"Missing Modality Robustness in Semi-Supervised Multi-Modal Semantic Segmentation","author":"Maheshwari","year":"2024"},{"key":"10.1016\/j.imavis.2026.105977_b41","series-title":"Proceedings of the 41st International Conference on Machine Learning","first-page":"21753","article-title":"GeminiFusion: Efficient pixel-wise multimodal fusion for vision transformer","volume":"vol. 235","author":"Jia","year":"2024"},{"key":"10.1016\/j.imavis.2026.105977_b42","series-title":"2024 Intelligent Methods, Systems, and Applications","first-page":"456","article-title":"Rhrsegnet: Relighting high-resolution night-time semantic segmentation","author":"Elmahdy","year":"2024"},{"key":"10.1016\/j.imavis.2026.105977_b43","doi-asserted-by":"crossref","DOI":"10.1016\/j.image.2025.117265","article-title":"Darksegnet: Low-light semantic segmentation network based on image pyramid","volume":"135","author":"Tan","year":"2025","journal-title":"Signal Process., Image Commun."},{"key":"10.1016\/j.imavis.2026.105977_b44","doi-asserted-by":"crossref","unstructured":"J. Zhou, X. Zhou, S. Chan, Z. Chen, X. Zhang, Enhancing nighttime semantic segmentation with visual-linguistic priors and wavelet transform, in: J. Kwok (Ed.), Proceedings of the Thirty-Fourth International Joint Conference on Artificial Intelligence, IJCAI-25, http:\/\/dx.doi.org\/10.24963\/ijcai.2025\/888.","DOI":"10.24963\/ijcai.2025\/888"},{"key":"10.1016\/j.imavis.2026.105977_b45","series-title":"2019 IEEE International Conference on Image Processing","first-page":"2996","article-title":"What\u2019s there in the dark","author":"Nag","year":"2019"},{"key":"10.1016\/j.imavis.2026.105977_b46","series-title":"Exploring reliable matching with phase enhancement for night-time semantic segmentation","author":"Pan","year":"2024"},{"key":"10.1016\/j.imavis.2026.105977_b47","doi-asserted-by":"crossref","DOI":"10.1016\/j.imavis.2024.105149","article-title":"Nighttime image semantic segmentation with retinex theory","volume":"148","author":"Sun","year":"2024","journal-title":"Image Vis. Comput."},{"issue":"2","key":"10.1016\/j.imavis.2026.105977_b48","first-page":"2443","article-title":"Ed-ged: Nighttime image semantic segmentation based on enhanced detail and bidirectional guidance","volume":"80","author":"Yuan","year":"2024","journal-title":"Comput. Mater. Contin."},{"key":"10.1016\/j.imavis.2026.105977_b49","series-title":"2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"16917","article-title":"Nightlab: A dual-level architecture with hardness detection for segmentation at night","author":"Deng","year":"2022"},{"key":"10.1016\/j.imavis.2026.105977_b50","doi-asserted-by":"crossref","DOI":"10.1016\/j.jvcir.2024.104116","article-title":"Liis: Low-light image instance segmentation","volume":"100","author":"Li","year":"2024","journal-title":"J. Vis. Commun. Image Represent."},{"key":"10.1016\/j.imavis.2026.105977_b51","series-title":"2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"15764","article-title":"DANNet: A one-stage domain adaptation network for unsupervised nighttime semantic segmentation","author":"Wu","year":"2021"},{"key":"10.1016\/j.imavis.2026.105977_b52","series-title":"2023 IEEE\/CVF International Conference on Computer Vision","first-page":"21515","article-title":"Cmda: Cross-modality domain adaptation for nighttime semantic segmentation","author":"Xia","year":"2023"},{"key":"10.1016\/j.imavis.2026.105977_b53","series-title":"2023 IEEE\/CVF International Conference on Computer Vision","first-page":"8070","article-title":"Similarity Min-Max: Zero-Shot Day-Night Domain Adaptation","author":"Luo","year":"2023"},{"key":"10.1016\/j.imavis.2026.105977_b54","doi-asserted-by":"crossref","DOI":"10.1007\/s00440-014-0576-6","article-title":"Reconstruction and estimation in the planted partition model","author":"Mossel","year":"2015","journal-title":"Probab. Theory Related Fields"},{"issue":"4","key":"10.1016\/j.imavis.2026.105977_b55","doi-asserted-by":"crossref","first-page":"6497","DOI":"10.1109\/LRA.2021.3093652","article-title":"Ms-uda: Multi-spectral unsupervised domain adaptation for thermal image semantic segmentation","volume":"6","author":"Kim","year":"2021","journal-title":"IEEE Robot. Autom. Lett."},{"key":"10.1016\/j.imavis.2026.105977_b56","series-title":"2020 IEEE International Conference on Robotics and Automation","first-page":"9441","article-title":"Pst900: Rgb-thermal calibration, dataset and segmentation network","author":"Shivakumar","year":"2020"},{"key":"10.1016\/j.imavis.2026.105977_b57","doi-asserted-by":"crossref","unstructured":"W. Ji, J. Li, C. Bian, Z. Zhang, L. Cheng, Semanticrt: A large-scale dataset and method for robust semantic segmentation in multispectral images, in: Proceedings of the 31th ACM International Conference on Multimedia, 2023, pp. 3307\u20133316.","DOI":"10.1145\/3581783.3611738"},{"issue":"2","key":"10.1016\/j.imavis.2026.105977_b58","doi-asserted-by":"crossref","first-page":"129","DOI":"10.1109\/TIT.1982.1056489","article-title":"Least squares quantization in pcm","volume":"28","author":"Lloyd","year":"1982","journal-title":"IEEE Trans. Inform. Theory"},{"issue":"1\u20132","key":"10.1016\/j.imavis.2026.105977_b59","doi-asserted-by":"crossref","first-page":"83","DOI":"10.1002\/nav.3800020109","article-title":"The hungarian method for the assignment problem","volume":"2","author":"Kuhn","year":"1955","journal-title":"Nav. Res. Logist. Q."},{"key":"10.1016\/j.imavis.2026.105977_b60","series-title":"2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"3637","article-title":"Unsupervised Semantic Segmentation Through Depth-Guided Feature Correlation and Sampling","author":"Sick","year":"2024"},{"key":"10.1016\/j.imavis.2026.105977_b61","unstructured":"S.F. Bhat, R. Birkl, D. Wofk, P. Wonka, M. M\u00fcller, Zoedepth: Zero-shot transfer by combining relative and metric depth, arXiv preprint arXiv:2302.12288 (Unpublished)."},{"key":"10.1016\/j.imavis.2026.105977_b62","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.111398","article-title":"Resolving semantic conflicts in rgb-t semantic segmentation","volume":"162","author":"Zhao","year":"2025","journal-title":"Pattern Recognit."},{"issue":"3","key":"10.1016\/j.imavis.2026.105977_b63","doi-asserted-by":"crossref","first-page":"2576","DOI":"10.1109\/LRA.2019.2904733","article-title":"Rtfnet: Rgb-thermal fusion network for semantic segmentation of urban scenes","volume":"4","author":"Sun","year":"2019","journal-title":"IEEE Robot. Autom. Lett."}],"container-title":["Image and Vision Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0262885626000843?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0262885626000843?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,4,27]],"date-time":"2026-04-27T11:20:15Z","timestamp":1777288815000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0262885626000843"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6]]},"references-count":63,"alternative-id":["S0262885626000843"],"URL":"https:\/\/doi.org\/10.1016\/j.imavis.2026.105977","relation":{},"ISSN":["0262-8856"],"issn-type":[{"value":"0262-8856","type":"print"}],"subject":[],"published":{"date-parts":[[2026,6]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Distilling auxiliary RGB\u2013T features for unsupervised semantic segmentation","name":"articletitle","label":"Article Title"},{"value":"Image and Vision Computing","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.imavis.2026.105977","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"105977"}}