{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T18:15:25Z","timestamp":1783188925853,"version":"3.54.6"},"reference-count":78,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100017550","name":"Shaanxi Science and Technology Association","doi-asserted-by":"publisher","award":["A024360001"],"award-info":[{"award-number":["A024360001"]}],"id":[{"id":"10.13039\/501100017550","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Neural Networks"],"published-print":{"date-parts":[[2026,12]]},"DOI":"10.1016\/j.neunet.2026.109187","type":"journal-article","created":{"date-parts":[[2026,5,28]],"date-time":"2026-05-28T15:04:20Z","timestamp":1779980660000},"page":"109187","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Semantic consistency-aware pseudo-temporal framework for multimodal remote sensing image segmentation"],"prefix":"10.1016","volume":"204","author":[{"given":"Yujia","family":"Sun","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuejiang","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Weisheng","family":"Dong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Le","family":"Dong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Peng","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lichao","family":"Mou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2067-2763","authenticated-orcid":false,"given":"Xin","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"1","key":"10.1016\/j.neunet.2026.109187_bib0001","doi-asserted-by":"crossref","first-page":"200","DOI":"10.1007\/s11036-020-01703-3","article-title":"Convolutional neural network for the semantic segmentation of remote sensing images","volume":"26","author":"Alam","year":"2021","journal-title":"Mobile Networks and Applications"},{"key":"10.1016\/j.neunet.2026.109187_bib0002","doi-asserted-by":"crossref","DOI":"10.3389\/fnbot.2024.1427786","article-title":"Multi-modal remote perception learning for object sensory data","volume":"18","author":"Almujally","year":"2024","journal-title":"Frontiers in Neurorobotics"},{"key":"10.1016\/j.neunet.2026.109187_bib0003","doi-asserted-by":"crossref","first-page":"20","DOI":"10.1016\/j.isprsjprs.2017.11.011","article-title":"Beyond RGB: Very high resolution urban remote sensing with multimodal deep networks","volume":"140","author":"Audebert","year":"2018","journal-title":"ISPRS Journal of Photogrammetry and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109187_bib0004","unstructured":"Chen, J., Lu, Y., Yu, Q., Luo, X., Adeli, E., Wang, Y., Lu, L., Yuille, A. L., & Zhou, Y. (2021). TransUNet: Transformers make strong encoders for medical image segmentation. arXiv: 2102.04306."},{"issue":"10","key":"10.1016\/j.neunet.2026.109187_bib0005","doi-asserted-by":"crossref","first-page":"18855","DOI":"10.1109\/TITS.2022.3161977","article-title":"Disparity-based multiscale fusion network for transportation detection","volume":"23","author":"Chen","year":"2022","journal-title":"IEEE Transactions on Intelligent Transportation Systems"},{"key":"10.1016\/j.neunet.2026.109187_bib0006","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2024.102311","article-title":"Region-based online selective examination for weakly supervised semantic segmentation","volume":"107","author":"Chen","year":"2024","journal-title":"Information Fusion"},{"key":"10.1016\/j.neunet.2026.109187_bib0007","series-title":"European conference on computer vision","first-page":"561","article-title":"Bi-directional cross-modality feature propagation with separation-and-aggregation gate for RGB-D semantic segmentation","author":"Chen","year":"2020"},{"issue":"5","key":"10.1016\/j.neunet.2026.109187_bib0008","doi-asserted-by":"crossref","first-page":"1229","DOI":"10.3390\/rs15051229","article-title":"Semantic segmentation of remote sensing imagery based on multiscale deformable CNN and denseCRF","volume":"15","author":"Cheng","year":"2023","journal-title":"Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109187_bib0009","article-title":"Multispectral remote sensing object detection via selective cross-modal interaction and aggregation","author":"Cui","year":"2025","journal-title":"Neural Networks"},{"key":"10.1016\/j.neunet.2026.109187_bib0010","doi-asserted-by":"crossref","first-page":"94","DOI":"10.1016\/j.isprsjprs.2020.01.013","article-title":"ResUNet-a: A deep learning framework for semantic segmentation of remotely sensed data","volume":"162","author":"Diakogiannis","year":"2020","journal-title":"ISPRS Journal of Photogrammetry and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109187_bib0011","first-page":"1","article-title":"Distilling segmenters from CNNs and transformers for remote sensing images\u2019 semantic segmentation","volume":"61","author":"Dong","year":"2023","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109187_bib0012","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"6202","article-title":"Slowfast networks for video recognition","author":"Feichtenhofer","year":"2019"},{"key":"10.1016\/j.neunet.2026.109187_bib0013","doi-asserted-by":"crossref","DOI":"10.1109\/TGRS.2025.3553478","article-title":"FTransDeepLab: Multimodal fusion transformer-based DeepLabv3+ for remote sensing semantic segmentation","volume":"63","author":"Feng","year":"2025","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109187_bib0014","article-title":"Domain-continual learning for multi-center anatomical detection via prompt-enhanced and densely-fused medSAM","author":"Gao","year":"2025","journal-title":"Information Fusion"},{"key":"10.1016\/j.neunet.2026.109187_bib0015","doi-asserted-by":"crossref","first-page":"5147","DOI":"10.1109\/TPAMI.2025.3649001","article-title":"Crossearth: Geospatial vision foundation model for domain generalizable remote sensing semantic segmentation","volume":"48","author":"Gong","year":"2025","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.neunet.2026.109187_bib0016","series-title":"Asian conference on computer vision","first-page":"213","article-title":"FuseNet: Incorporating depth into semantic segmentation via fusion-based CNN architecture","author":"Hazirbas","year":"2016"},{"key":"10.1016\/j.neunet.2026.109187_bib0017","doi-asserted-by":"crossref","first-page":"1474","DOI":"10.1109\/TIP.2023.3245324","article-title":"Multimodal remote sensing image segmentation with intuition-inspired hypergraph modeling","volume":"32","author":"He","year":"2023","journal-title":"IEEE Transactions on Image Processing"},{"issue":"3","key":"10.1016\/j.neunet.2026.109187_bib0018","doi-asserted-by":"crossref","first-page":"722","DOI":"10.3390\/math11030722","article-title":"MFTransNet: A multi-modal fusion with CNN-transformer network for semantic segmentation of HSR remote sensing images","volume":"11","author":"He","year":"2023","journal-title":"Mathematics"},{"key":"10.1016\/j.neunet.2026.109187_bib0019","first-page":"1","article-title":"Swin transformer embedding UNet for remote sensing image semantic segmentation","volume":"60","author":"He","year":"2022","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109187_bib0020","doi-asserted-by":"crossref","first-page":"96","DOI":"10.1016\/j.isprsjprs.2021.12.007","article-title":"CMGFNet: A deep cross-modal gated fusion network for building extraction from very high-resolution remote sensing images","volume":"184","author":"Hosseinpour","year":"2022","journal-title":"ISPRS Journal of Photogrammetry and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109187_bib0021","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"7132","article-title":"Squeeze-and-excitation networks","author":"Hu","year":"2018"},{"key":"10.1016\/j.neunet.2026.109187_bib0022","first-page":"1","article-title":"Boundary enhancement semantic segmentation for building extraction from remote sensed image","volume":"60","author":"Jung","year":"2021","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109187_bib0023","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"4015","article-title":"Segment anything","author":"Kirillov","year":"2023"},{"key":"10.1016\/j.neunet.2026.109187_bib0024","series-title":"Proceedings of the 31st ACM SIGKDD conference on knowledge discovery and data mining","first-page":"1330","article-title":"FusionSAM: Visual multi-modal learning with segment anything model","volume":"2","author":"Li","year":"2025"},{"issue":"1","key":"10.1016\/j.neunet.2026.109187_bib0025","article-title":"A review of remote sensing image segmentation by deep learning methods","volume":"17","author":"Li","year":"2024","journal-title":"International Journal of Digital Earth"},{"key":"10.1016\/j.neunet.2026.109187_bib0026","article-title":"Semi-medSAM: Adapting SAM-assisted semi-supervised multi-modality learning for medical endoscopic image segmentation","author":"Li","year":"2026","journal-title":"Pattern Recognition"},{"key":"10.1016\/j.neunet.2026.109187_bib0027","doi-asserted-by":"crossref","first-page":"3905","DOI":"10.1109\/JSTARS.2025.3527213","article-title":"LMF-Net: A learnable multi-modal fusion network for semantic segmentation of remote sensing data","volume":"10","author":"Li","year":"2025","journal-title":"IEEE Journal of Selected Topics in Applied Earth Observations and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109187_bib0028","first-page":"1","article-title":"Multistage attention resu-net for semantic segmentation of fine-resolution remote sensing images","volume":"19","author":"Li","year":"2022","journal-title":"IEEE Geoscience and Remote Sensing Letters"},{"key":"10.1016\/j.neunet.2026.109187_bib0029","doi-asserted-by":"crossref","first-page":"84","DOI":"10.1016\/j.isprsjprs.2021.09.005","article-title":"ABCNet: Attentive bilateral contextual network for efficient semantic segmentation of fine-resolution remotely sensed imagery","volume":"181","author":"Li","year":"2021","journal-title":"ISPRS Journal of Photogrammetry and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109187_bib0030","article-title":"RefSAM: Efficiently adapting segmenting anything model for referring video object segmentation","author":"Li","year":"2025","journal-title":"Neural Networks"},{"issue":"9","key":"10.1016\/j.neunet.2026.109187_bib0031","doi-asserted-by":"crossref","first-page":"7871","DOI":"10.1109\/TGRS.2020.3034123","article-title":"AFNet: Adaptive fusion network for remote sensing image semantic segmentation","volume":"59","author":"Liu","year":"2020","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109187_bib0032","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2024.102352","article-title":"A semantic-driven coupled network for infrared and visible image fusion","volume":"108","author":"Liu","year":"2024","journal-title":"Information Fusion"},{"key":"10.1016\/j.neunet.2026.109187_bib0033","first-page":"1","article-title":"Rethinking transformers for semantic segmentation of remote sensing images","volume":"61","author":"Liu","year":"2023","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109187_bib0034","doi-asserted-by":"crossref","first-page":"8175","DOI":"10.1109\/OJCOMS.2025.3608296","article-title":"On the computational efficiency of fading models: The fluctuating two-ray model for mmwave bands","volume":"6","author":"L\u00f3pez-Ben\u00edtez","year":"2025","journal-title":"IEEE Open Journal of the Communications Society"},{"key":"10.1016\/j.neunet.2026.109187_bib0035","doi-asserted-by":"crossref","DOI":"10.1109\/TGRS.2024.3458446","article-title":"CTCFNet: CNN-Transformer complementary and fusion network for high-resolution remote sensing image semantic segmentation","volume":"62","author":"Lu","year":"2024","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"issue":"2","key":"10.1016\/j.neunet.2026.109187_bib0036","doi-asserted-by":"crossref","first-page":"172","DOI":"10.1016\/j.inpa.2023.02.001","article-title":"Semantic segmentation of agricultural images: A survey","volume":"11","author":"Luo","year":"2024","journal-title":"Information Processing in Agriculture"},{"issue":"1","key":"10.1016\/j.neunet.2026.109187_bib0037","doi-asserted-by":"crossref","first-page":"654","DOI":"10.1038\/s41467-024-44824-z","article-title":"Segment anything in medical images","volume":"15","author":"Ma","year":"2024","journal-title":"Nature Communications"},{"key":"10.1016\/j.neunet.2026.109187_bib0038","doi-asserted-by":"crossref","first-page":"3463","DOI":"10.1109\/JSTARS.2022.3165005","article-title":"A crossmodal multiscale fusion network for semantic segmentation of remote sensing data","volume":"15","author":"Ma","year":"2022","journal-title":"IEEE Journal of Selected Topics in Applied Earth Observations and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109187_bib0039","first-page":"1","article-title":"RS3Mamba: Visual state space model for remote sensing image semantic segmentation","volume":"21","author":"Ma","year":"2024","journal-title":"IEEE Geoscience and Remote Sensing Letters"},{"key":"10.1016\/j.neunet.2026.109187_bib0040","first-page":"5405015","article-title":"A unified framework with multimodal fine-tuning for remote sensing semantic segmentation","volume":"33","author":"Ma","year":"2025","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109187_bib0041","first-page":"1","article-title":"A multilevel multimodal fusion transformer for remote sensing semantic segmentation","volume":"62","author":"Ma","year":"2024","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109187_bib0042","first-page":"1","article-title":"A multilevel multimodal fusion transformer for remote sensing semantic segmentation","volume":"62","author":"Ma","year":"2024","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109187_bib0043","doi-asserted-by":"crossref","first-page":"385","DOI":"10.1016\/j.isprsjprs.2020.07.005","article-title":"Fully convolutional networks for land cover classification from historical panchromatic aerial photographs","volume":"167","author":"Mboga","year":"2020","journal-title":"ISPRS Journal of Photogrammetry and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109187_bib0044","doi-asserted-by":"crossref","first-page":"152","DOI":"10.1016\/j.isprsjprs.2013.11.001","article-title":"Contextual classification of lidar data and building object detection in urban areas","volume":"87","author":"Niemeyer","year":"2014","journal-title":"ISPRS Journal of Photogrammetry and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109187_bib0045","article-title":"LBMS-SAM: Segment anything model guided sem image segmentation for lithium battery materials","author":"Qi","year":"2025","journal-title":"Neural Networks"},{"key":"10.1016\/j.neunet.2026.109187_bib0046","series-title":"2021\u202fIEEE International conference on robotics and automation (ICRA)","first-page":"13525","article-title":"Efficient RGB-d semantic segmentation for indoor scene analysis","author":"Seichter","year":"2021"},{"key":"10.1016\/j.neunet.2026.109187_bib0047","unstructured":"Sherrah, J. (2016). Fully convolutional networks for dense semantic labelling of high-resolution aerial imagery. arXiv: 1606.02585."},{"key":"10.1016\/j.neunet.2026.109187_bib0048","doi-asserted-by":"crossref","DOI":"10.1016\/j.neunet.2025.107414","article-title":"Lunetr: Language-infused unetr for precise pancreatic tumor segmentation in 3d medical image","volume":"187","author":"Shi","year":"2025","journal-title":"Neural Networks"},{"issue":"11","key":"10.1016\/j.neunet.2026.109187_bib0049","doi-asserted-by":"crossref","first-page":"16687","DOI":"10.1109\/TITS.2024.3409874","article-title":"Subjective driving risk prediction based on spatiotemporal distribution features of human driver\u2019s cognitive risk","volume":"25","author":"Song","year":"2024","journal-title":"IEEE Transactions on Intelligent Transportation Systems"},{"key":"10.1016\/j.neunet.2026.109187_bib0050","doi-asserted-by":"crossref","DOI":"10.1016\/j.autcon.2025.106484","article-title":"Training-free automatic instance segmentation of girder bridge point cloud via large model fusion with reverse entity modelling verification","volume":"179","author":"Song","year":"2025","journal-title":"Automation in Construction"},{"key":"10.1016\/j.neunet.2026.109187_bib0051","doi-asserted-by":"crossref","first-page":"11797","DOI":"10.1109\/TCSVT.2025.3579580","article-title":"Distilling hierarchical knowledge from multimodal fusion for unimodal image segmentation","volume":"35","author":"Sun","year":"2025","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.neunet.2026.109187_bib0052","first-page":"1","article-title":"Deep multimodal fusion network for semantic segmentation using remote sensing image and LiDAR data","volume":"60","author":"Sun","year":"2021","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109187_bib0053","unstructured":"Talaei, S., Daneshfar, F., Abdullah, A. A., & Khan, M. (2026). BiCLIP: Bidirectional and consistent language-image processing for robust medical image segmentation. arXiv: 2603.00156."},{"key":"10.1016\/j.neunet.2026.109187_bib0054","doi-asserted-by":"crossref","first-page":"5115","DOI":"10.1109\/TIP.2025.3595408","article-title":"Multi-scale autoencoder suppression strategy for hyperspectral image anomaly detection","volume":"34","author":"Tu","year":"2025","journal-title":"IEEE Transactions on Image Processing"},{"key":"10.1016\/j.neunet.2026.109187_bib0055","first-page":"1","article-title":"Building extraction with vision transformer","volume":"60","author":"Wang","year":"2022","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109187_bib0056","doi-asserted-by":"crossref","first-page":"196","DOI":"10.1016\/j.isprsjprs.2022.06.008","article-title":"UNetFormer: A UNet-like transformer for efficient semantic segmentation of remote sensing urban scene imagery","volume":"190","author":"Wang","year":"2022","journal-title":"ISPRS Journal of Photogrammetry and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109187_bib0057","first-page":"1","article-title":"Multisenseseg: A cost-effective unified multimodal semantic segmentation model for remote sensing","volume":"62","author":"Wang","year":"2024","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109187_bib0058","doi-asserted-by":"crossref","first-page":"5257","DOI":"10.1109\/TIP.2022.3192706","article-title":"Extendable multiple nodes recurrent tracking framework with RTU++","volume":"31","author":"Wang","year":"2022","journal-title":"IEEE Transactions on Image Processing"},{"issue":"4","key":"10.1016\/j.neunet.2026.109187_bib0059","doi-asserted-by":"crossref","DOI":"10.1002\/aisy.202200131","article-title":"Tasta: Text-assisted spatial and temporal attention network for video question answering","volume":"5","author":"Wang","year":"2023","journal-title":"Advanced Intelligent Systems"},{"key":"10.1016\/j.neunet.2026.109187_bib0060","first-page":"1","article-title":"A vit-based multiscale feature fusion approach for remote sensing image segmentation","volume":"19","author":"Wang","year":"2022","journal-title":"IEEE Geoscience and Remote Sensing Letters"},{"key":"10.1016\/j.neunet.2026.109187_bib0061","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2025.103030","article-title":"Peafusion: Parameter-efficient adaptation for rgb-thermal fusion-based semantic segmentation","volume":"120","author":"Wang","year":"2025","journal-title":"Information Fusion"},{"key":"10.1016\/j.neunet.2026.109187_bib0062","article-title":"Combining feature compensation and GCN-based reconstruction for multimodal remote sensing image semantic segmentation","author":"Wang","year":"2025","journal-title":"Information Fusion"},{"key":"10.1016\/j.neunet.2026.109187_bib0063","first-page":"1","article-title":"Cmtfnet: CNN and multiscale transformer fusion network for remote-sensing image semantic segmentation","volume":"61","author":"Wu","year":"2023","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109187_bib0064","doi-asserted-by":"crossref","first-page":"2213","DOI":"10.1109\/TIP.2024.3374070","article-title":"Toward video anomaly retrieval from video anomaly detection: New benchmarks and model","volume":"33","author":"Wu","year":"2024","journal-title":"IEEE Transactions on Image Processing"},{"key":"10.1016\/j.neunet.2026.109187_bib0065","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2026.104222","article-title":"Pia: Fusing edge prior information into attention for semantic segmentation in vision transformer","author":"Xiao","year":"2026","journal-title":"Information Fusion"},{"key":"10.1016\/j.neunet.2026.109187_bib0066","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2025.103338","article-title":"Samamba: Adaptive state space modeling with hierarchical vision for infrared small target detection","author":"Xu","year":"2025","journal-title":"Information Fusion"},{"key":"10.1016\/j.neunet.2026.109187_bib0067","series-title":"2021\u202fIEEE International conference on robotics and biomimetics (ROBIO)","first-page":"1129","article-title":"NLFNet: Non-local fusion towards generalized multimodal semantic segmentation across RGB-depth, polarization, and thermal images","author":"Yan","year":"2021"},{"key":"10.1016\/j.neunet.2026.109187_bib0068","doi-asserted-by":"crossref","first-page":"3023","DOI":"10.1109\/JSTARS.2024.3349657","article-title":"Ssnet: A novel transformer and CNN hybrid network for remote sensing semantic segmentation","volume":"17","author":"Yao","year":"2024","journal-title":"IEEE Journal of Selected Topics in Applied Earth Observations and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109187_bib0069","doi-asserted-by":"crossref","DOI":"10.1016\/j.neunet.2024.106622","article-title":"Dark-DSAR: Lightweight one-step pipeline for action recognition in dark videos","volume":"179","author":"Yin","year":"2024","journal-title":"Neural Networks"},{"key":"10.1016\/j.neunet.2026.109187_bib0070","series-title":"European conference on computer vision","first-page":"18","article-title":"Pseudo-ris: Distinctive pseudo-supervision generation for referring image segmentation","author":"Yu","year":"2024"},{"issue":"1","key":"10.1016\/j.neunet.2026.109187_bib0071","doi-asserted-by":"crossref","first-page":"16","DOI":"10.1109\/TGRS.2012.2234755","article-title":"Remote sensing image segmentation by combining spectral and texture features","volume":"52","author":"Yuan","year":"2013","journal-title":"IEEE Transactions on geoscience and remote sensing"},{"key":"10.1016\/j.neunet.2026.109187_bib0072","doi-asserted-by":"crossref","first-page":"280","DOI":"10.1016\/j.isprsjprs.2020.09.025","article-title":"Identifying and mapping individual plants in a highly diverse high-elevation ecosystem using UAV imagery and deep learning","volume":"169","author":"Zhang","year":"2020","journal-title":"ISPRS Journal of Photogrammetry and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109187_bib0073","article-title":"A novel spatial-temporal learning method for enhancing generalization in adaptive video streaming","author":"Zhang","year":"2025","journal-title":"IEEE Transactions on Mobile Computing"},{"key":"10.1016\/j.neunet.2026.109187_bib0074","doi-asserted-by":"crossref","DOI":"10.1016\/j.neunet.2025.107881","article-title":"SS-KAN: Self-supervised Kolmogorov-Arnold networks for limited data remote sensing semantic segmentation","author":"Zhang","year":"2025","journal-title":"Neural Networks"},{"issue":"11","key":"10.1016\/j.neunet.2026.109187_bib0075","doi-asserted-by":"crossref","first-page":"1755","DOI":"10.1109\/LGRS.2018.2857804","article-title":"Hyperspectral unmixing via deep convolutional neural networks","volume":"15","author":"Zhang","year":"2018","journal-title":"IEEE Geoscience and Remote Sensing Letters"},{"key":"10.1016\/j.neunet.2026.109187_bib0076","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"2881","article-title":"Pyramid scene parsing network","author":"Zhao","year":"2017"},{"key":"10.1016\/j.neunet.2026.109187_bib0077","first-page":"1","article-title":"A self-learning-update CNN model for semantic segmentation of remote sensing images","volume":"20","author":"Zheng","year":"2023","journal-title":"IEEE Geoscience and Remote Sensing Letters"},{"key":"10.1016\/j.neunet.2026.109187_bib0078","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"4096","article-title":"Foreground-aware relation network for geospatial object segmentation in high spatial resolution remote sensing imagery","author":"Zheng","year":"2020"}],"container-title":["Neural Networks"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0893608026006489?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0893608026006489?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T17:17:56Z","timestamp":1783185476000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0893608026006489"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,12]]},"references-count":78,"alternative-id":["S0893608026006489"],"URL":"https:\/\/doi.org\/10.1016\/j.neunet.2026.109187","relation":{},"ISSN":["0893-6080"],"issn-type":[{"value":"0893-6080","type":"print"}],"subject":[],"published":{"date-parts":[[2026,12]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Semantic consistency-aware pseudo-temporal framework for multimodal remote sensing image segmentation","name":"articletitle","label":"Article Title"},{"value":"Neural Networks","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.neunet.2026.109187","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"109187"}}