{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T10:20:30Z","timestamp":1777630830901,"version":"3.51.4"},"reference-count":43,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100013804","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100013804","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Information Sciences"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1016\/j.ins.2026.123298","type":"journal-article","created":{"date-parts":[[2026,2,26]],"date-time":"2026-02-26T16:00:00Z","timestamp":1772121600000},"page":"123298","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["MLaVQA: A multi-level attention method for remote sensing visual question answering with large language model"],"prefix":"10.1016","volume":"742","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7933-6946","authenticated-orcid":false,"given":"Weipeng","family":"Jing","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-9579-1936","authenticated-orcid":false,"given":"Wanlin","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1932-7698","authenticated-orcid":false,"given":"Chao","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1290-4272","authenticated-orcid":false,"given":"Mahmoud","family":"Emam","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.ins.2026.123298_bib0005","doi-asserted-by":"crossref","DOI":"10.1016\/j.cities.2025.106122","article-title":"Urban safety perception assessments via integrating multimodal large language models with street view images","volume":"165","author":"Zhang","year":"2025","journal-title":"Cities"},{"key":"10.1016\/j.ins.2026.123298_bib0010","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"5481","article-title":"EarthVQA: towards queryable earth via relational reasoning-based remote sensing visual question answering","volume":"vol. 38","author":"Wang","year":"2024"},{"key":"10.1016\/j.ins.2026.123298_bib0015","doi-asserted-by":"crossref","DOI":"10.1016\/j.engappai.2025.110759","article-title":"Visual geo-localization and attitude estimation using satellite imagery and topographical elevation for unmanned aerial vehicles","volume":"153","author":"Qiu","year":"2025","journal-title":"Eng. Appl. Artif. Intell."},{"key":"10.1016\/j.ins.2026.123298_bib0020","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2023.102192","article-title":"NCGLF2: network combining global and local features for fusion of multisource remote sensing data","volume":"104","author":"Tu","year":"2024","journal-title":"Inf. Fusion."},{"key":"10.1016\/j.ins.2026.123298_bib0025","doi-asserted-by":"crossref","first-page":"32","DOI":"10.1109\/MGRS.2024.3383473","article-title":"Vision-language models in remote sensing: current progress and future trends","volume":"12","author":"Li","year":"2024","journal-title":"IEEE Geosci. Remote Sens. Mag."},{"key":"10.1016\/j.ins.2026.123298_bib0030","series-title":"ICLR","article-title":"MogaNet: multi-order gated aggregation network","author":"Li","year":"2024"},{"key":"10.1016\/j.ins.2026.123298_bib0035","doi-asserted-by":"crossref","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","article-title":"Long short-term memory","volume":"9","author":"Hochreiter","year":"1997","journal-title":"Neural Comput."},{"key":"10.1016\/j.ins.2026.123298_bib0040","doi-asserted-by":"crossref","first-page":"595","DOI":"10.1109\/TII.2019.2934144","article-title":"A temporally irreversible visual attention model inspired by motion sensitive neurons","volume":"16","author":"Xu","year":"2020","journal-title":"IEEE Trans. Ind. Informat."},{"key":"10.1016\/j.ins.2026.123298_bib0045","doi-asserted-by":"crossref","first-page":"8555","DOI":"10.1109\/TGRS.2020.2988782","article-title":"RSVQA: visual question answering for remote sensing data","volume":"58","author":"Lobry","year":"2020","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.ins.2026.123298_bib0050","series-title":"Proceedings of the 36th International Conference on Machine Learning, Volume 97 of Proceedings of Machine Learning Research, PMLR","first-page":"6105","article-title":"EfficientNet: rethinking model scaling for convolutional neural networks","author":"Tan","year":"2019"},{"key":"10.1016\/j.ins.2026.123298_bib0055","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV)","first-page":"568","article-title":"Pyramid vision transformer: a versatile backbone for dense prediction without convolutions","author":"Wang","year":"2021"},{"key":"10.1016\/j.ins.2026.123298_bib0060","doi-asserted-by":"crossref","first-page":"2581","DOI":"10.1007\/s11263-024-02289-z","article-title":"Clip-powered tass: target-aware single-stream network for audio-visual question answering","volume":"133","author":"Jiang","year":"2025","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.ins.2026.123298_bib0065","series-title":"Proc Conf Assoc Comput Linguist Meet 2019","first-page":"6558","article-title":"Multimodal transformer for unaligned multimodal language sequences","author":"Liang","year":"2019"},{"key":"10.1016\/j.ins.2026.123298_bib0070","series-title":"Proceedings of the 40th International Conference on Machine Learning, Volume 202 of Proceedings of Machine Learning Research, PMLR","first-page":"19730","article-title":"BLIP-2: bootstrapping language-image pre-training with frozen image encoders and large language models","author":"Li","year":"2023"},{"key":"10.1016\/j.ins.2026.123298_bib0075","first-page":"2729","article-title":"PAL-BERT: an improved question answering model","volume":"139","author":"Zheng","year":"2024","journal-title":"CMES - Comput. Model. Eng. Sci."},{"key":"10.1016\/j.ins.2026.123298_bib0080","first-page":"771","article-title":"DPAL-BERT: a faster and lighter question answering model","volume":"141","author":"Yin","year":"2024","journal-title":"CMES - Comput. Model. Eng. Sci."},{"key":"10.1016\/j.ins.2026.123298_bib0085","doi-asserted-by":"crossref","first-page":"422","DOI":"10.1016\/j.isprsjprs.2024.05.001","article-title":"EarthVQANet: multi-task visual question answering for remote sensing image understanding","volume":"212","author":"Wang","year":"2024","journal-title":"ISPRS J. Photogramm. Remote Sens."},{"key":"10.1016\/j.ins.2026.123298_bib0090","doi-asserted-by":"crossref","DOI":"10.1002\/aisy.202200131","article-title":"Tasta: text-assisted spatial and temporal attention network for video question answering","volume":"5","author":"Wang","year":"2023","journal-title":"Adv. Intell. Syst."},{"key":"10.1016\/j.ins.2026.123298_bib0095","doi-asserted-by":"crossref","DOI":"10.3389\/fnbot.2024.1427786","article-title":"Multi-modal remote perception learning for object sensory data","volume":"18","author":"Almujally","year":"2024","journal-title":"Front. Neurorobotics"},{"key":"10.1016\/j.ins.2026.123298_bib0100","doi-asserted-by":"crossref","DOI":"10.1016\/j.engappai.2024.109660","article-title":"Multilingual entity alignment by abductive knowledge reasoning on multiple knowledge graphs","volume":"139","author":"Akhtar","year":"2025","journal-title":"Eng. Appl. Artif. Intell."},{"key":"10.1016\/j.ins.2026.123298_bib0105","doi-asserted-by":"crossref","DOI":"10.1155\/2023\/9429505","article-title":"GDF: a novel image fusion approach for compelling depiction of earthly features","author":"Mishra","year":"2023","journal-title":"J. Sens."},{"key":"10.1016\/j.ins.2026.123298_bib0110","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","article-title":"Momentum contrast for unsupervised visual representation learning","author":"He","year":"2020"},{"key":"10.1016\/j.ins.2026.123298_bib0115","series-title":"Proceedings of the 38th International Conference on Machine Learning, Volume 139 of Proceedings of Machine Learning Research, PMLR","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.ins.2026.123298_bib0120","doi-asserted-by":"crossref","first-page":"717","DOI":"10.1109\/TIP.2025.3650045","article-title":"IAP: improving continual learning of vision-language models via instance-aware prompting","volume":"35","author":"Fu","year":"2026","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.ins.2026.123298_bib0125","first-page":"1","article-title":"S2DBFT: spectral\u2013spatial dual-branch fusion transformer for hyperspectral image classification","volume":"63","author":"Yiheng","year":"2025","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.ins.2026.123298_bib0130","series-title":"2025 IEEE 22nd International Symposium on Biomedical Imaging (ISBI)","first-page":"1","article-title":"Conditional visuo-textual prompt learning for medical image analysis","author":"Mineo","year":"2025"},{"key":"10.1016\/j.ins.2026.123298_bib0135","doi-asserted-by":"crossref","DOI":"10.1016\/j.rse.2024.114386","article-title":"Deep learning for retrieving omni-directional ocean wave spectra from spaceborne synthetic aperture radar","volume":"314","author":"Wu","year":"2024","journal-title":"Remote Sens. Environ."},{"key":"10.1016\/j.ins.2026.123298_bib0140","series-title":"Proceedings of the Thirty-Third International Joint Conference on Artificial Intelligence, IJCAI \u201924","article-title":"Lemevit: efficient vision transformer with learnable meta tokens for remote sensing image interpretation","author":"Jiang","year":"2024"},{"key":"10.1016\/j.ins.2026.123298_bib0145","series-title":"Computer Vision \u2013 ECCV 2024","first-page":"164","article-title":"MMEarth: exploring multi-modal pretext tasks for geospatial representation learning","author":"Nedungadi","year":"2025"},{"key":"10.1016\/j.ins.2026.123298_bib0150","first-page":"5805","article-title":"Skyscript: a large and semantically diverse vision-language dataset for remote sensing","volume":"38","author":"Wang","year":"2024","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"10.1016\/j.ins.2026.123298_bib0155","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"28202","article-title":"Osprey: pixel understanding with visual instruction tuning","author":"Yuan","year":"2024"},{"key":"10.1016\/j.ins.2026.123298_bib0160","first-page":"1","article-title":"Zero-shot automatic modulation recognition using a large vision-language model","author":"Zhao","year":"2025","journal-title":"IEEE Trans. Commun."},{"key":"10.1016\/j.ins.2026.123298_bib0165","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"27831","article-title":"GeoChat: grounded large vision-language model for remote sensing","author":"Kuckreja","year":"2024"},{"key":"10.1016\/j.ins.2026.123298_bib0170","first-page":"1","article-title":"EarthGPT: a universal multimodal large language model for multisensor image comprehension in remote sensing domain","volume":"62","author":"Zhang","year":"2024","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.ins.2026.123298_bib0175","doi-asserted-by":"crossref","first-page":"179","DOI":"10.1038\/s41612-025-01070-4","article-title":"AirGPT: pioneering the convergence of conversational AI with atmospheric science","volume":"8","author":"Song","year":"2025","journal-title":"npj Clim. Atmos. Sci."},{"key":"10.1016\/j.ins.2026.123298_bib0180","series-title":"Proceedings of the Eighteenth ACM International Conference on Web Search and Data Mining, WSDM \u201925","first-page":"261","article-title":"Large language model simulator for cold-start recommendation","author":"Huang","year":"2025"},{"key":"10.1016\/j.ins.2026.123298_bib0185","author":"Touvron"},{"key":"10.1016\/j.ins.2026.123298_bib0190","doi-asserted-by":"crossref","first-page":"92","DOI":"10.1016\/j.neucom.2022.06.111","article-title":"Activation functions in deep learning: a comprehensive survey and benchmark","volume":"503","author":"Dubey","year":"2022","journal-title":"Neurocomputing"},{"key":"10.1016\/j.ins.2026.123298_bib0195","series-title":"Thirty-Fifth Conference on Neural Information Processing Systems Datasets and Benchmarks Track (Round 2)","article-title":"LoveDA: a remote sensing land-cover dataset for domain adaptive semantic segmentation","author":"Wang","year":"2021"},{"key":"10.1016\/j.ins.2026.123298_bib0200","series-title":"Proceedings of the 37th International Conference on Machine Learning","first-page":"1597","article-title":"A simple framework for contrastive learning of visual representations","volume":"vol. 119","author":"Chen","year":"2020"},{"key":"10.1016\/j.ins.2026.123298_bib0205","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV)","first-page":"5785","article-title":"FastVIT: a fast hybrid vision transformer using structural reparameterization","author":"Vasu","year":"2023"},{"key":"10.1016\/j.ins.2026.123298_bib0210","article-title":"Scenario-guided temporal prototypes in reinforcement learning","volume":"8","author":"Dobravec","year":"2026","journal-title":"Mach. Learn. Knowl. Extr."},{"key":"10.1016\/j.ins.2026.123298_bib0215","doi-asserted-by":"crossref","DOI":"10.3390\/technologies14010035","article-title":"Neurostrainsense: a transformer-generative AI framework for stress detection using heterogeneous multimodal datasets","volume":"14","author":"Ben Ismail","year":"2026","journal-title":"Technologies"}],"container-title":["Information Sciences"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S002002552600229X?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S002002552600229X?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T03:10:04Z","timestamp":1777432204000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S002002552600229X"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6]]},"references-count":43,"alternative-id":["S002002552600229X"],"URL":"https:\/\/doi.org\/10.1016\/j.ins.2026.123298","relation":{},"ISSN":["0020-0255"],"issn-type":[{"value":"0020-0255","type":"print"}],"subject":[],"published":{"date-parts":[[2026,6]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"MLaVQA: A multi-level attention method for remote sensing visual question answering with large language model","name":"articletitle","label":"Article Title"},{"value":"Information Sciences","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.ins.2026.123298","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Inc. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"123298"}}