{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,10]],"date-time":"2026-06-10T18:02:16Z","timestamp":1781114536478,"version":"3.54.1"},"reference-count":50,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by-nc\/4.0\/"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,11]]},"DOI":"10.1016\/j.patcog.2026.113874","type":"journal-article","created":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T16:25:11Z","timestamp":1777479911000},"page":"113874","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PD","title":["GeoRL: Adaptive tokenization via reinforcement learning for remote sensing foundation models"],"prefix":"10.1016","volume":"179","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4420-4158","authenticated-orcid":false,"given":"Hengyang","family":"He","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6930-8674","authenticated-orcid":false,"given":"Le","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ce","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.patcog.2026.113874_b1","first-page":"1","article-title":"RSPrompter: Learning to prompt for remote sensing instance segmentation based on visual foundation model","volume":"62","author":"Chen","year":"2024","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.patcog.2026.113874_b2","article-title":"DINOv2: Learning robust visual features without supervision","author":"Oquab","year":"2023","journal-title":"Trans. Mach. Learn. Res."},{"key":"10.1016\/j.patcog.2026.113874_b3","first-page":"1","article-title":"RemoteCLIP: A vision language foundation model for remote sensing","volume":"62","author":"Liu","year":"2024","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.patcog.2026.113874_b4","first-page":"1","article-title":"EarthGPT: A universal multimodal large language model for multi-sensor image comprehension in remote sensing domain","volume":"62","author":"Zhang","year":"2024","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"issue":"11","key":"10.1016\/j.patcog.2026.113874_b5","doi-asserted-by":"crossref","first-page":"7778","DOI":"10.1109\/TPAMI.2021.3117983","article-title":"Object detection in aerial images: A large-scale benchmark and challenges","volume":"44","author":"Ding","year":"2022","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.patcog.2026.113874_b6","doi-asserted-by":"crossref","first-page":"2372","DOI":"10.1109\/JSTARS.2023.3347595","article-title":"Dual encoder\u2013decoder network for land cover segmentation of remote sensing image","volume":"17","author":"Wang","year":"2024","journal-title":"IEEE J. Sel. Top. Appl. Earth Obs. Remote. Sens."},{"key":"10.1016\/j.patcog.2026.113874_b7","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110983","article-title":"MSNet: Multi-scale network for object detection in remote sensing images","volume":"158","author":"Gao","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113874_b8","unstructured":"A. Dosovitskiy, L. Beyer, A. Kolesnikov, et al., An image is worth 16x16 words: Transformers for image recognition at scale, in: International Conference on Learning Representations, ICLR, 2021."},{"key":"10.1016\/j.patcog.2026.113874_b9","unstructured":"H. Touvron, M. Cord, M. Douze, F. Massa, A. Sablayrolges, H. J\u00e9gou, Training data-efficient image transformers & distillation through attention, in: International Conference on Machine Learning, ICML, 2021, pp. 10347\u201310357."},{"key":"10.1016\/j.patcog.2026.113874_b10","doi-asserted-by":"crossref","unstructured":"Z. Liu, Y. Lin, Y. Cao, H. Hu, Y. Wei, Z. Zhang, S. Lin, B. Guo, Swin Transformer: Hierarchical vision transformer using shifted windows, in: IEEE\/CVF International Conference on Computer Vision, ICCV, 2021, pp. 10012\u201310022.","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"10.1016\/j.patcog.2026.113874_b11","unstructured":"H. Bao, L. Dong, S. Piao, F. Wei, BEiT: BERT pre-training of image transformers, in: International Conference on Learning Representations, ICLR, 2022."},{"key":"10.1016\/j.patcog.2026.113874_b12","article-title":"RingMo: A remote sensing foundation model with masked image modeling","volume":"61","author":"Sun","year":"2023","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.patcog.2026.113874_b13","first-page":"1","article-title":"A multilevel multimodal fusion transformer for remote sensing semantic segmentation","volume":"62","author":"Ma","year":"2024","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"issue":"2","key":"10.1016\/j.patcog.2026.113874_b14","doi-asserted-by":"crossref","first-page":"262","DOI":"10.1080\/10095020.2022.2085633","article-title":"Deep learning for change detection in remote sensing: A review","volume":"26","author":"Bai","year":"2023","journal-title":"Geo-Spatial Inf. Sci."},{"key":"10.1016\/j.patcog.2026.113874_b15","first-page":"3185","article-title":"LSKNet: A foundation lightweight backbone for remote sensing","volume":"132","author":"Li","year":"2025","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.patcog.2026.113874_b16","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.111478","article-title":"Optical remote sensing image salient object detection via bidirectional cross-attention and attention restoration","volume":"164","author":"Gu","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113874_b17","first-page":"1","article-title":"Multiattention network for semantic segmentation of fine-resolution remote sensing images","volume":"60","author":"Li","year":"2022","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.patcog.2026.113874_b18","series-title":"European Conference on Computer Vision","first-page":"396","article-title":"Adaptive token sampling for efficient vision transformers","author":"Fayyaz","year":"2022"},{"key":"10.1016\/j.patcog.2026.113874_b19","first-page":"1","article-title":"MSANet: Multiscale self-attention aggregation network for few-shot aerial imagery segmentation","volume":"62","author":"Li","year":"2024","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.patcog.2026.113874_b20","doi-asserted-by":"crossref","first-page":"24","DOI":"10.1016\/j.patrec.2025.03.035","article-title":"STFormer: An efficient visual transformer model with sparse attention and adaptive token aggregation","volume":"196","author":"Feng","year":"2025","journal-title":"Pattern Recognit. Lett."},{"key":"10.1016\/j.patcog.2026.113874_b21","first-page":"1","article-title":"RockFormer: A U-shaped transformer network for martian rock segmentation","volume":"61","author":"Liu","year":"2023","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.patcog.2026.113874_b22","series-title":"IEEE International Conference on Image Processing","first-page":"2313","article-title":"Density-guided dense pseudo label selection for semi-supervised oriented object detection","author":"Zhao","year":"2024"},{"key":"10.1016\/j.patcog.2026.113874_b23","first-page":"1","article-title":"Enhancing multiscale representations with transformer for remote sensing image semantic segmentation","volume":"61","author":"Xiao","year":"2023","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.patcog.2026.113874_b24","series-title":"Language-guided token compression with reinforcement learning in large vision-language models","author":"Cao","year":"2026"},{"key":"10.1016\/j.patcog.2026.113874_b25","doi-asserted-by":"crossref","DOI":"10.1016\/j.engappai.2023.107610","article-title":"Passable area segmentation for open-pit mine road from vehicle perspective","volume":"129","author":"Zheng","year":"2024","journal-title":"Eng. Appl. Artif. Intell."},{"issue":"7540","key":"10.1016\/j.patcog.2026.113874_b26","doi-asserted-by":"crossref","first-page":"529","DOI":"10.1038\/nature14236","article-title":"Human-level control through deep reinforcement learning","volume":"518","author":"Mnih","year":"2015","journal-title":"Nature"},{"issue":"8","key":"10.1016\/j.patcog.2026.113874_b27","doi-asserted-by":"crossref","first-page":"5329","DOI":"10.1109\/TGRS.2019.2899057","article-title":"Classification of hyperspectral images based on multiclass spatial-spectral generative adversarial networks","volume":"57","author":"Feng","year":"2019","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"issue":"2","key":"10.1016\/j.patcog.2026.113874_b28","doi-asserted-by":"crossref","first-page":"305","DOI":"10.3390\/rs14020305","article-title":"Superpixel-based attention graph neural network for semantic segmentation in aerial images","volume":"14","author":"Diao","year":"2022","journal-title":"Remote. Sens."},{"key":"10.1016\/j.patcog.2026.113874_b29","article-title":"Recurrent progressive fusion-based learning for multi-source remote sensing image classification","volume":"159","author":"Zhang","year":"2026","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113874_b30","doi-asserted-by":"crossref","first-page":"2733","DOI":"10.1007\/s10462-021-10061-9","article-title":"Deep reinforcement learning in computer vision: A comprehensive survey","volume":"55","author":"Le","year":"2022","journal-title":"Artif. Intell. Rev."},{"key":"10.1016\/j.patcog.2026.113874_b31","first-page":"1","article-title":"Domain-invariant progressive knowledge distillation for UAV-based object detection","volume":"22","author":"Yao","year":"2025","journal-title":"IEEE Geosci. Remote. Sens. Lett."},{"issue":"4","key":"10.1016\/j.patcog.2026.113874_b32","first-page":"4605","article-title":"Glance and focus networks for dynamic visual recognition","volume":"45","author":"Huang","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.patcog.2026.113874_b33","unstructured":"Y. Liang, C. Ge, Z. Tong, Y. Song, J. Wang, P. Xie, Not all patches are what you need: Expediting vision transformers via token reorganizations, in: International Conference on Learning Representations, ICLR, 2022."},{"key":"10.1016\/j.patcog.2026.113874_b34","series-title":"International Conference on Learning Representations","article-title":"Token merging: Your ViT but faster","author":"Bolya","year":"2023"},{"key":"10.1016\/j.patcog.2026.113874_b35","doi-asserted-by":"crossref","unstructured":"M. Chen, W. Shao, P. Xu, M. Lin, K. Zhang, F. Chao, R. Ji, Y. Qiao, P. Luo, DiffRate: Differentiable compression rate for efficient vision transformers, in: IEEE\/CVF International Conference on Computer Vision, ICCV, 2023, pp. 17139\u201317148.","DOI":"10.1109\/ICCV51070.2023.01574"},{"key":"10.1016\/j.patcog.2026.113874_b36","series-title":"GeoReason: Aligning thinking and answering in remote sensing vision-language models via logical consistency reinforcement learning","author":"Li","year":"2026"},{"key":"10.1016\/j.patcog.2026.113874_b37","doi-asserted-by":"crossref","first-page":"94","DOI":"10.1016\/j.neucom.2023.02.008","article-title":"Improving proximal policy optimization with alpha divergence","volume":"534","author":"Xu","year":"2023","journal-title":"Neurocomputing"},{"key":"10.1016\/j.patcog.2026.113874_b38","article-title":"SatMAE: Pre-training transformers for temporal and multi-spectral satellite imagery","volume":"vol. 35","author":"Cong","year":"2022"},{"issue":"4","key":"10.1016\/j.patcog.2026.113874_b39","doi-asserted-by":"crossref","first-page":"1452","DOI":"10.1109\/TPAMI.2020.2974745","article-title":"Gliding vertex on the horizontal bounding box for multi-oriented object detection","volume":"43","author":"Xu","year":"2021","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.patcog.2026.113874_b40","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110976","article-title":"TBNet: A texture and boundary-aware network for small weak object detection in remote-sensing imagery","volume":"158","author":"Li","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113874_b41","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.111503","article-title":"Progressive class-aware instance enhancement for aircraft detection in remote sensing imagery","volume":"164","author":"Shi","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113874_b42","series-title":"European Conference on Computer Vision","first-page":"161","article-title":"Projecting points to axes: Oriented object detection via point-axis representation","author":"Zhao","year":"2024"},{"key":"10.1016\/j.patcog.2026.113874_b43","doi-asserted-by":"crossref","first-page":"196","DOI":"10.1016\/j.isprsjprs.2022.06.008","article-title":"UNetFormer: A UNet-like transformer for efficient semantic segmentation of remote sensing urban scene imagery","volume":"190","author":"Wang","year":"2022","journal-title":"ISPRS J. Photogramm. Remote Sens."},{"key":"10.1016\/j.patcog.2026.113874_b44","first-page":"1","article-title":"RRSIS: Referring remote sensing image segmentation","volume":"62","author":"Yuan","year":"2024","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.patcog.2026.113874_b45","doi-asserted-by":"crossref","first-page":"51","DOI":"10.1016\/j.isprsjprs.2024.01.022","article-title":"HD-Net: High-resolution decoupled network for building footprint extraction via deeply supervised body and boundary decomposition","volume":"209","author":"Li","year":"2024","journal-title":"ISPRS J. Photogramm. Remote Sens."},{"key":"10.1016\/j.patcog.2026.113874_b46","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.111583","article-title":"Layerlink: Bridging remote sensing object detection and large vision models with efficient fine-tuning","volume":"165","author":"Zhu","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113874_b47","doi-asserted-by":"crossref","first-page":"738","DOI":"10.1109\/JSTARS.2022.3230835","article-title":"Vision transformer with contrastive learning for remote sensing image scene classification","volume":"16","author":"Bi","year":"2023","journal-title":"IEEE J. Sel. Top. Appl. Earth Obs. Remote. Sens."},{"key":"10.1016\/j.patcog.2026.113874_b48","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110959","article-title":"Multimodal self-supervised learning for remote sensing data land cover classification","volume":"157","author":"Xue","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113874_b49","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1109\/LGRS.2023.3239263","article-title":"Local window attention transformer for polarimetric SAR image classification","volume":"20","author":"Jamali","year":"2023","journal-title":"IEEE Geosci. Remote. Sens. Lett."},{"key":"10.1016\/j.patcog.2026.113874_b50","first-page":"5765","article-title":"SkyScript: A large and semantically diverse vision-language dataset for remote sensing","volume":"vol. 38","author":"Wang","year":"2024"}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326008393?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326008393?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,10]],"date-time":"2026-06-10T17:15:24Z","timestamp":1781111724000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0031320326008393"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,11]]},"references-count":50,"alternative-id":["S0031320326008393"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113874","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,11]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"GeoRL: Adaptive tokenization via reinforcement learning for remote sensing foundation models","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113874","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 The Authors. Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"113874"}}