{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T07:10:25Z","timestamp":1783926625009,"version":"3.55.0"},"reference-count":70,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Neural Networks"],"published-print":{"date-parts":[[2027,1]]},"DOI":"10.1016\/j.neunet.2026.109308","type":"journal-article","created":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T23:26:46Z","timestamp":1783034806000},"page":"109308","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PA","title":["Towards robust remote sensing visual question answering with spectral expert adaptation and group-relative optimization"],"prefix":"10.1016","volume":"205","author":[{"given":"Junjiang","family":"Yuan","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhe","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tingbin","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.neunet.2026.109308_bib0001","unstructured":"Achiam, J., Adler, S., Agarwal, S., Ahmad, L., Akkaya, I., Aleman, F. L., Almeida, D., Altenschmidt, J., Altman, S., Anadkat, S. et al. (2023). GPT-4 technical report. arXiv preprint arXiv: 2303.08774."},{"key":"10.1016\/j.neunet.2026.109308_bib0002","doi-asserted-by":"crossref","unstructured":"Aghajanyan, A., Zettlemoyer, L., & Gupta, S. (2021). Intrinsic Dimensionality Explains the Effectiveness of Language Model Fine-Tuning. In Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long papers)(pp.7319\u20137328). Association for Computational Linguistics. 10.18653\/v1\/2021.acl-long.568.","DOI":"10.18653\/v1\/2021.acl-long.568"},{"key":"10.1016\/j.neunet.2026.109308_bib0003","series-title":"Proceedings of the 62nd annual meeting of the association for computational linguistics (volume 1: Long papers)","first-page":"12248","article-title":"Back to basics: Revisiting REINFORCE-style optimization for learning from human feedback in LLMs","author":"Ahmadian","year":"2024"},{"key":"10.1016\/j.neunet.2026.109308_bib0004","doi-asserted-by":"crossref","first-page":"23716","DOI":"10.52202\/068431-1723","article-title":"Flamingo: A visual language model for few-shot learning","volume":"35","author":"Alayrac","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.neunet.2026.109308_bib0005","unstructured":"Bai, J., Bai, S., Yang, S., Wang, S., Tan, S., Wang, P., Lin, J., Zhou, C., & Zhou, J. (2023). Qwen-VL: A frontier large vision-language model with versatile abilities. arXiv preprint arXiv: 2308.12966."},{"key":"10.1016\/j.neunet.2026.109308_bib0006","unstructured":"Bai, S., Cai, Y., Chen, R., Chen, K., Chen, X., Cheng, Z., Deng, L., Ding, W., Gao, C., Ge, C., Ge, W., Guo, Z., Huang, Q., Huang, J., Huang, F., Hui, B., Jiang, S., Li, Z., Li, M., Li, M., Li, K., Lin, Z., Lin, J., Liu, X., Liu, J., Liu, C., Liu, Y., Liu, D., Liu, S., Lu, D., Luo, R., Lv, C., Men, R., Meng, L., Ren, X., Ren, X., Song, S., Sun, Y., Tang, J., Tu, J., Wan, J., Wang, P., Wang, P., Wang, Q., Wang, Y., Xie, T., Xu, Y., Xu, H., Xu, J., Yang, Z., Yang, M., Yang, J., Yang, A., Yu, B., Zhang, F., Zhang, H., Zhang, X., Zheng, B., Zhong, H., Zhou, J., Zhou, F., Zhou, J., Zhu, Y., & Zhu, K. (2025). Qwen3-VL technical report. arXiv preprint arXiv: 2511.21631."},{"key":"10.1016\/j.neunet.2026.109308_bib0007","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1109\/TGRS.2022.3192460","article-title":"Bi-modal transformer-based approach for visual question answering in remote sensing imagery","volume":"60","author":"Bazi","year":"2022","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109308_bib0008","unstructured":"Bleeker, M., Hendriksen, M., Yates, A., & de Rijke, M. (2024). Demonstrating and reducing shortcuts in vision-language representation learning. arXiv preprint arXiv: 2402.17510."},{"key":"10.1016\/j.neunet.2026.109308_bib0009","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"1372","article-title":"Prompt-RSVQA: Prompting visual context to a language model for remote sensing visual question answering","author":"Chappuis","year":"2022"},{"key":"10.1016\/j.neunet.2026.109308_bib0010","unstructured":"Chen, H., Tu, H., Wang, F., Liu, H., Tang, X., Du, X., Zhou, Y., & Xie, C. (2025). SFT or RL? An early investigation into training R1-like reasoning large vision-language models. arXiv preprint arXiv: 2504.11468."},{"key":"10.1016\/j.neunet.2026.109308_bib0011","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"18030","article-title":"VisualGPT: Data-efficient adaptation of pretrained language models for image captioning","author":"Chen","year":"2022"},{"key":"10.1016\/j.neunet.2026.109308_bib0012","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"24185","article-title":"InternVL: Scaling up vision foundation models and aligning for generic visual-linguistic tasks","author":"Chen","year":"2024"},{"key":"10.1016\/j.neunet.2026.109308_bib0013","unstructured":"Deng, H., Zou, D., Ma, R., Luo, H., Cao, Y., & Kang, Y. (2025a). Boosting the generalization and reasoning of vision language models with curriculum reinforcement learning. arXiv preprint arXiv: 2503.07065."},{"key":"10.1016\/j.neunet.2026.109308_bib0014","unstructured":"Deng, Y., Bansal, H., Yin, F., Peng, N., Wang, W., & Chang, K.-W. (2025b). OpenVLThinker: An early exploration to complex vision-language reasoning via iterative self-improvement. arXiv preprint arXiv: 2503.17352."},{"key":"10.1016\/j.neunet.2026.109308_bib0015","doi-asserted-by":"crossref","unstructured":"Dettmers, T., Pagnoni, A., Holtzman, A., & Zettlemoyer, L. (2023). QLoRA: Efficient finetuning of quantized LLMs. Advances in Neural Information Processing Systems, 36, 10088\u201310115. Curran Associates, Inc.","DOI":"10.52202\/075280-0441"},{"key":"10.1016\/j.neunet.2026.109308_bib0016","unstructured":"Guan, J., Mei, H., Zhang, B., Liu, D., Fu, Y., & Zhang, Y. (2025). UAV-VL-R1: Generalizing vision-language models via supervised fine-tuning and multi-stage GRPO for UAV visual reasoning. arXiv preprint arXiv: 2508.11196."},{"key":"10.1016\/j.neunet.2026.109308_bib0017","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"14375","article-title":"HallusionBench: An advanced diagnostic suite for entangled language hallucination and visual illusion in large vision-language models","author":"Guan","year":"2024"},{"key":"10.1016\/j.neunet.2026.109308_bib0018","unstructured":"Guo, D., Yang, D., Zhang, H., Song, J., Zhang, R., Xu, R., Zhu, Q., Ma, S., Wang, P., Bi, X. et al. (2025). DeepSeek-R1: Incentivizing reasoning capability in llms via reinforcement learning. arXiv preprint arXiv: 2501.12948."},{"key":"10.1016\/j.neunet.2026.109308_bib0019","series-title":"IGARSS 2023-2023 IEEE international geoscience and remote sensing symposium","first-page":"2231","article-title":"Lit-4-RSVQA: Lightweight transformer-based visual question answering in remote sensing","author":"Hackel","year":"2023"},{"key":"10.1016\/j.neunet.2026.109308_bib0020","unstructured":"Hayou, S., Ghosh, N., & Yu, B. (2024). LoRA+: Efficient low rank adaptation of large models. In Proceedings of the 41st International Conference on Machine Learning, Proceedings of Machine Learning ResearchPMLR. 235(pp. 17783\u201317806). https:\/\/proceedings.mlr.press\/v235\/hayou24a.html."},{"key":"10.1016\/j.neunet.2026.109308_bib0021","series-title":"International conference on learning representations","article-title":"Towards a unified view of parameter-efficient transfer learning","author":"He","year":"2021"},{"key":"10.1016\/j.neunet.2026.109308_bib0022","series-title":"International conference on machine learning","first-page":"2790","article-title":"Parameter-efficient transfer learning for NLP","author":"Houlsby","year":"2019"},{"key":"10.1016\/j.neunet.2026.109308_bib0023","unstructured":"Hu, E. J., Shen, Y., Wallis, P., Allen-Zhu, Z., Li, Y., Wang, S., Wang, L., & Chen, W. (2021). LoRA: Low-rank adaptation of large language models. arXiv preprint arXiv: 2106.09685."},{"key":"10.1016\/j.neunet.2026.109308_bib0024","unstructured":"Hu, Y., Yuan, J., Wen, C., Lu, X., & Li, X. (2023). RSGPT: A remote sensing vision language model and benchmark. arXiv preprint arXiv: 2307.15266."},{"key":"10.1016\/j.neunet.2026.109308_bib0025","unstructured":"Huang, W., Jia, B., Zhai, Z., Cao, S., Ye, Z., Zhao, F., Xu, Z., Hu, Y., & Lin, S. (2025). Vision-R1: Incentivizing reasoning capability in multimodal large language models. arXiv preprint arXiv: 2503.06749."},{"key":"10.1016\/j.neunet.2026.109308_bib0026","unstructured":"Irvin, J. A., Liu, E. R., Chen, J. C., Dormoy, I., Kim, J., Khanna, S., Zheng, Z., & Ermon, S. (2024). TEOChat: A large vision-language assistant for temporal earth observation data. arXiv preprint arXiv: 2410.06234."},{"key":"10.1016\/j.neunet.2026.109308_bib0027","unstructured":"Jaech, A., Kalai, A., Lerer, A., Richardson, A., El-Kishky, A., Low, A., Helyar, A., Madry, A., Beutel, A., Carney, A. et al. (2024). OpenAI o1 system card. arXiv preprint arXiv: 2412.16720."},{"key":"10.1016\/j.neunet.2026.109308_bib0028","series-title":"Proceedings of the 2021 conference on empirical methods in natural language processing","article-title":"The power of scale for parameter-efficient prompt tuning","author":"Lester","year":"2021"},{"key":"10.1016\/j.neunet.2026.109308_bib0029","unstructured":"Li, L. H., Yatskar, M., Yin, D., Hsieh, C.-J., & Chang, K.-W. (2019). VisualBERT: A simple and performant baseline for vision and language. arXiv preprint arXiv: 1908.03557."},{"key":"10.1016\/j.neunet.2026.109308_bib0030","series-title":"Proceedings of the 59th annual meeting of the association for computational linguistics and the 11th international joint conference on natural language processing (volume 1: Long papers)","first-page":"4582","article-title":"Prefix-tuning: Optimizing continuous prompts for generation","author":"Li","year":"2021"},{"key":"10.1016\/j.neunet.2026.109308_bib0031","doi-asserted-by":"crossref","unstructured":"Li, Z., Wu, X., Shi, G., Qin, Y., Du, H., Zhou, T., Manocha, D., & Boyd-Graber, J. L. (2025). VideoHallu: Evaluating and mitigating multi-modal hallucinations on synthetic video understanding. arXiv preprint arXiv: 2505.01481.","DOI":"10.32388\/BXC6X1"},{"key":"10.1016\/j.neunet.2026.109308_bib0032","unstructured":"Liu, F., Lin, K., Li, L., Wang, J., Yacoob, Y., & Wang, L. (2023a). Mitigating hallucination in large multi-modal models via robust instruction tuning. arXiv preprint arXiv: 2306.14565."},{"key":"10.1016\/j.neunet.2026.109308_bib0033","doi-asserted-by":"crossref","first-page":"34892","DOI":"10.52202\/075280-1516","article-title":"Visual instruction tuning","volume":"36","author":"Liu","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.neunet.2026.109308_bib0034","unstructured":"Liu, S.-Y., Wang, C.-Y., Yin, H., Molchanov, P., Wang, Y.-C. F., Cheng, K.-T., & Chen, M.-H. (2024). DoRA: Weight-decomposed low-rank adaptation. In Proceedings of the 41st International Conference on Machine Learning, Proceedings of Machine Learning ResearchPMLR. 235(pp. 32100\u201332121). https:\/\/proceedings.mlr.press\/v235\/liu24bn.html."},{"key":"10.1016\/j.neunet.2026.109308_bib0035","series-title":"2021\u202fIEEE International geoscience and remote sensing symposium IGARSS","first-page":"1218","article-title":"RSVQA meets bigearthnet: A new, large-scale, visual question answering dataset for remote sensing","author":"Lobry","year":"2021"},{"issue":"12","key":"10.1016\/j.neunet.2026.109308_bib0036","doi-asserted-by":"crossref","first-page":"8555","DOI":"10.1109\/TGRS.2020.2988782","article-title":"RSVQA: Visual question answering for remote sensing data","volume":"58","author":"Lobry","year":"2020","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109308_bib0037","unstructured":"Luo, G., Huang, M., Zhou, Y., Sun, X., Jiang, G., Wang, Z., & Ji, R. (2023). Towards efficient visual adaption via structural re-parameterization. arXiv preprint arXiv: 2302.08106."},{"key":"10.1016\/j.neunet.2026.109308_bib0038","series-title":"Proceedings of the 60th annual meeting of the association for computational linguistics (volume 1: Long papers)","first-page":"6253","article-title":"UniPELT: A unified framework for parameter-efficient language model tuning","author":"Mao","year":"2022"},{"key":"10.1016\/j.neunet.2026.109308_bib0039","unstructured":"Peng, Y., Zhang, G., Zhang, M., You, Z., Liu, J., Zhu, Q., Yang, K., Xu, X., Geng, X., & Yang, X. (2025). LMM-R1: Empowering 3B LMMs with strong reasoning abilities through two-stage rule-based Rl. arXiv preprint arXiv: 2503.07536."},{"key":"10.1016\/j.neunet.2026.109308_bib0040","series-title":"Proc. international conference on machine learning (ICML)","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.neunet.2026.109308_bib0041","doi-asserted-by":"crossref","first-page":"53728","DOI":"10.52202\/075280-2338","article-title":"Direct preference optimization: Your language model is secretly a reward model","volume":"36","author":"Rafailov","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.neunet.2026.109308_bib0042","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., & Klimov, O. (2017). Proximal policy optimization algorithms. arXiv preprint arXiv: 1707.06347."},{"key":"10.1016\/j.neunet.2026.109308_bib0043","unstructured":"Shao, Z., Wang, P., Zhu, Q., Xu, R., Song, J., Bi, X., Zhang, H., Zhang, M., Li, Y. K., Wu, Y. et al. (2024). DeepSeekMath: Pushing the limits of mathematical reasoning in open language models. arXiv preprint arXiv: 2402.03300."},{"key":"10.1016\/j.neunet.2026.109308_bib0044","unstructured":"Shen, H., Liu, P., Li, J., Fang, C., Ma, Y., Liao, J., Shen, Q., Zhang, Z., Zhao, K., Zhang, Q. et al. (2025). VLM-R1: A stable and generalizable R1-style large vision-language model. arXiv preprint arXiv: 2504.07615."},{"key":"10.1016\/j.neunet.2026.109308_bib0045","series-title":"Proceedings of the computer vision and pattern recognition conference","first-page":"29958","article-title":"R-TPT: Improving adversarial robustness of vision-language models through test-time prompt tuning","author":"Sheng","year":"2025"},{"key":"10.1016\/j.neunet.2026.109308_bib0046","series-title":"Image and signal processing for remote sensing XXVIII","first-page":"162","article-title":"Multi-modal fusion transformer for visual question answering in remote sensing","volume":"vol. 12267","author":"Siebert","year":"2022"},{"key":"10.1016\/j.neunet.2026.109308_bib0047","series-title":"Proceedings of the computer vision and pattern recognition conference","first-page":"14303","article-title":"EarthDial: Turning multi-sensory earth observations to interactive dialogues","author":"Soni","year":"2025"},{"key":"10.1016\/j.neunet.2026.109308_bib0048","unstructured":"Tan, H., Ji, Y., Hao, X., Lin, M., Wang, P., Wang, Z., & Zhang, S. (2025). Reason-RFT: Reinforcement fine-tuning for visual reasoning. arXiv preprint arXiv: 2503.20752."},{"key":"10.1016\/j.neunet.2026.109308_bib0049","doi-asserted-by":"crossref","unstructured":"Valipour, M., Rezagholizadeh, M., Kobyzev, I., & Ghodsi, A. (2023). DyloRA: Parameter efficient tuning of pre-trained models using dynamic search-free low-rank adaptation. In Proceedings of the 17th Conference of the European Chapter of the Association for Computational Linguistics, Association for Computational Linguistics. (pp. 3274\u20133287). 10.18653\/v1\/2023.eacl-main.239.","DOI":"10.18653\/v1\/2023.eacl-main.239"},{"key":"10.1016\/j.neunet.2026.109308_bib0050","first-page":"1","article-title":"RSAdapter: Adapting multimodal models for remote sensing visual question answering","volume":"62","author":"Wang","year":"2024","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109308_bib0051","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"3974","article-title":"DOTA: A large-scale dataset for object detection in aerial images","author":"Xia","year":"2018"},{"issue":"7","key":"10.1016\/j.neunet.2026.109308_bib0052","doi-asserted-by":"crossref","first-page":"3965","DOI":"10.1109\/TGRS.2017.2685945","article-title":"AID: A benchmark data set for performance evaluation of aerial scene classification","volume":"55","author":"Xia","year":"2017","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109308_bib0053","unstructured":"Xia, J., Zang, Y., Gao, P., Li, Y., & Zhou, K. (2025). Visionary-R1: Mitigating shortcuts in visual reasoning with reinforcement learning. arXiv preprint arXiv: 2505.14677."},{"key":"10.1016\/j.neunet.2026.109308_bib0054","doi-asserted-by":"crossref","DOI":"10.1007\/s44443-026-00605-w","article-title":"A lightweight model for indoor object detection in unstructured scenes based on joint attention and prior knowledge","author":"Xing","year":"2026","journal-title":"Journal of King Saud University \u2013 Computer and Information Sciences"},{"key":"10.1016\/j.neunet.2026.109308_bib0055","doi-asserted-by":"crossref","DOI":"10.1016\/j.measurement.2025.118794","article-title":"A human\u2013computer interactive rehabilitation system for non-contact gesture recognition based on 3d graph deep learning","volume":"257","author":"Xing","year":"2026","journal-title":"Measurement"},{"key":"10.1016\/j.neunet.2026.109308_bib0056","series-title":"Proceedings of the 18th SIGSPATIAL international conference on advances in geographic information systems","first-page":"270","article-title":"Bag-of-visual-words and spatial extensions for land-use classification","author":"Yang","year":"2010"},{"key":"10.1016\/j.neunet.2026.109308_bib0057","first-page":"1","article-title":"From easy to hard: Learning language-guided curriculum for visual question answering on remote sensing data","volume":"60","author":"Yuan","year":"2022","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109308_bib0058","series-title":"Proceedings of the 60th annual meeting of the association for computational linguistics (volume 2: Short papers)","first-page":"1","article-title":"BitFit: Simple parameter-efficient fine-tuning for transformer-based masked language-models","author":"Zaken","year":"2022"},{"key":"10.1016\/j.neunet.2026.109308_bib0059","doi-asserted-by":"crossref","unstructured":"Zhan, Y., Xiong, Z., & Yuan, Y. (2024). SkyEyeGPT: Unifying remote sensing vision-language tasks via instruction tuning with large language model. arXiv preprint arXiv: 2401.09712.","DOI":"10.1016\/j.isprsjprs.2025.01.020"},{"issue":"4","key":"10.1016\/j.neunet.2026.109308_bib0060","doi-asserted-by":"crossref","first-page":"2175","DOI":"10.1109\/TGRS.2014.2357078","article-title":"Saliency-guided unsupervised feature learning for scene classification","volume":"53","author":"Zhang","year":"2014","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109308_bib0061","unstructured":"Zhang, F., & Pilanci, M. (2024a). Riemannian preconditioned loRA for fine-tuning foundation models."},{"key":"10.1016\/j.neunet.2026.109308_bib0062","series-title":"The thirty-eighth annual conference on neural information processing systems","article-title":"Spectral adapter: Fine-tuning in spectral space","author":"Zhang","year":"2024"},{"key":"10.1016\/j.neunet.2026.109308_bib0063","unstructured":"Zhang, F. F., Li, L., Chen, J.-C., Jiang, Z., Wang, B., & Qian, Y. (2023a). IncreloRA: Incremental parameter allocation method for parameter-efficient fine-tuning. arXiv: 2308.12043, https:\/\/api.semanticscholar.org\/CorpusID:261076438."},{"key":"10.1016\/j.neunet.2026.109308_bib0064","doi-asserted-by":"crossref","unstructured":"Zhang, J., Huang, J., Yao, H., Liu, S., Zhang, X., Lu, S., & Tao, D. (2025). R1-VL: Learning to reason with multimodal large language models via step-wise group relative policy optimization. arXiv preprint arXiv: 2503.12937.","DOI":"10.1109\/ICCV51701.2025.00181"},{"key":"10.1016\/j.neunet.2026.109308_bib0065","unstructured":"Zhang, Q., Chen, M., Bukharin, A., Karampatziakis, N., He, P., Cheng, Y., Chen, W., & Zhao, T. (2023b). AdaloRA: Adaptive budget allocation for parameter-efficient fine- tuning. International Conference on Learning Representations, https:\/\/openreview.net\/forum?id=lq62uWRJjiY."},{"issue":"8","key":"10.1016\/j.neunet.2026.109308_bib0066","doi-asserted-by":"crossref","first-page":"5535","DOI":"10.1109\/TGRS.2019.2900302","article-title":"Hierarchical and robust convolutional neural network for very high-resolution remote sensing object detection","volume":"57","author":"Zhang","year":"2019","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109308_bib0067","first-page":"1","article-title":"A spatial hierarchical reasoning network for remote sensing visual question answering","volume":"61","author":"Zhang","year":"2023","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109308_bib0068","series-title":"Proc. of the AAAI","first-page":"29733","article-title":"SWIFT: A scalable lightweight infrastructure for fine-tuning","volume":"vol. 39","author":"Zhao","year":"2025"},{"key":"10.1016\/j.neunet.2026.109308_bib0069","first-page":"1","article-title":"Mutual attention inception network for remote sensing visual question answering","volume":"60","author":"Zheng","year":"2021","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.109308_bib0070","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"15659","article-title":"Prompt-aligned gradient for prompt tuning","author":"Zhu","year":"2023"}],"container-title":["Neural Networks"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0893608026007689?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0893608026007689?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T06:44:28Z","timestamp":1783925068000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0893608026007689"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2027,1]]},"references-count":70,"alternative-id":["S0893608026007689"],"URL":"https:\/\/doi.org\/10.1016\/j.neunet.2026.109308","relation":{},"ISSN":["0893-6080"],"issn-type":[{"value":"0893-6080","type":"print"}],"subject":[],"published":{"date-parts":[[2027,1]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Towards robust remote sensing visual question answering with spectral expert adaptation and group-relative optimization","name":"articletitle","label":"Article Title"},{"value":"Neural Networks","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.neunet.2026.109308","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"109308"}}