{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T19:51:54Z","timestamp":1783972314873,"version":"3.55.0"},"reference-count":46,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100007085","name":"National University of Defense Technology","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100007085","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100010097","name":"China Association for Science and Technology","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100010097","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.patcog.2026.113365","type":"journal-article","created":{"date-parts":[[2026,2,28]],"date-time":"2026-02-28T15:59:56Z","timestamp":1772294396000},"page":"113365","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":6,"special_numbering":"C","title":["GeoNav: Empowering MLLMs with dual-scale geospatial reasoning for language-goal aerial navigation"],"prefix":"10.1016","volume":"177","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5018-8687","authenticated-orcid":false,"given":"Haotian","family":"Xu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8115-7020","authenticated-orcid":false,"given":"Yue","family":"Hu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7561-5646","authenticated-orcid":false,"given":"Chen","family":"Gao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5805-834X","authenticated-orcid":false,"given":"Zhengqiu","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-3624-9114","authenticated-orcid":false,"given":"Yong","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1207-8660","authenticated-orcid":false,"given":"Quanjun","family":"Yin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.patcog.2026.113365_bib0001","series-title":"IEEE\/CVF International Conference on Computer Vision, ICCV 2023, Paris, France, October 1\u20136, 2023","first-page":"15338","article-title":"AerialVLN: vision-and-language navigation for UAVs","author":"Liu","year":"2023"},{"key":"10.1016\/j.patcog.2026.113365_bib0002","doi-asserted-by":"crossref","first-page":"6272","DOI":"10.1109\/LRA.2025.3566604","article-title":"UEVAVD: a dataset for developing UAV\u2019s eye view active object detection","volume":"10","author":"Jiang","year":"2024","journal-title":"IEEE Rob. Autom. Lett."},{"key":"10.1016\/j.patcog.2026.113365_bib0003","series-title":"2025IEEE International Conference on Robotics and Automation (ICRA)","first-page":"16202","article-title":"OpenBench: a new benchmark and baseline for semantic navigation in smart logistics","author":"Wang","year":"2025"},{"issue":"7","key":"10.1016\/j.patcog.2026.113365_bib0004","doi-asserted-by":"crossref","first-page":"6276","DOI":"10.1109\/TITS.2023.3343713","article-title":"AERIAL: a meta review and discussion of challenges toward unmanned aerial vehicle operations in logistics, mobility, and monitoring","volume":"25","author":"Wandelt","year":"2023","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"10.1016\/j.patcog.2026.113365_bib0005","series-title":"2nd CoRL Workshop on Learning Effective Abstractions for Planning","article-title":"Say-REAPEx: an LLM-Modulo UAV online planning framework for search and rescue","author":"D\u00f6schl","year":"2024"},{"key":"10.1016\/j.patcog.2026.113365_bib0006","doi-asserted-by":"crossref","first-page":"9502","DOI":"10.1109\/LRA.2025.3592098","article-title":"NEUSIS: a compositional neuro-symbolic framework for autonomous perception, reasoning, and planning in complex uav search missions","volume":"10","author":"Cai","year":"2025","journal-title":"IEEE Rob. Autom. Lett."},{"key":"10.1016\/j.patcog.2026.113365_bib0007","series-title":"The Fourteenth International Conference on Learning Representations","article-title":"OpenFly: a versatile toolchain and large-scale benchmark for aerial vision-language navigation","author":"Gao","year":"2026"},{"key":"10.1016\/j.patcog.2026.113365_bib0008","doi-asserted-by":"crossref","first-page":"4351","DOI":"10.52202\/075280-0193","article-title":"Frequency-enhanced data augmentation for vision-and-language navigation","volume":"36","author":"He","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.113365_bib0009","series-title":"Conference on Robot Learning","first-page":"492","article-title":"LM-NAV: robotic navigation with large pre-trained models of language, vision, and action","author":"Shah","year":"2023"},{"key":"10.1016\/j.patcog.2026.113365_bib0010","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"18924","article-title":"VELMA: verbalization embodiment of llm agents for vision and language navigation in street view","volume":"38","author":"Schumann","year":"2024"},{"key":"10.1016\/j.patcog.2026.113365_bib0011","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"5912","article-title":"CityNav: a large-scale dataset for real-world aerial navigation","author":"Lee","year":"2025"},{"key":"10.1016\/j.patcog.2026.113365_bib0012","first-page":"104934","article-title":"GOMAA-GEO: goal modality agnostic active geo-localization","volume":"37","author":"Sarkar","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.113365_bib0013","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"10707","article-title":"UrBench: a comprehensive benchmark for evaluating large multimodal models in multi-view urban scenarios","volume":"39","author":"Zhou","year":"2025"},{"issue":"9","key":"10.1016\/j.patcog.2026.113365_bib0014","doi-asserted-by":"crossref","first-page":"748","DOI":"10.1038\/s43588-023-00503-5","article-title":"Spatial planning of urban communities via deep reinforcement learning","volume":"3","author":"Zheng","year":"2023","journal-title":"Nat. Comput. Sci."},{"key":"10.1016\/j.patcog.2026.113365_bib0015","doi-asserted-by":"crossref","first-page":"256","DOI":"10.1016\/j.isprsjprs.2013.10.004","article-title":"Results of the ISPRS benchmark on urban object detection and 3D building reconstruction","volume":"93","author":"Rottensteiner","year":"2014","journal-title":"ISPRS J. Photogramm. Remote Sens."},{"key":"10.1016\/j.patcog.2026.113365_bib0016","unstructured":"G. Zhao, G. Li, J. Pan, Y. Yu, Aerial Vision-and-Language Navigation with Grid-based View Selection and Map Construction, (2025). arXiv: 2503.11091."},{"issue":"11","key":"10.1016\/j.patcog.2026.113365_bib0017","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3570723","article-title":"Path-planning for unmanned aerial vehicles with environment complexity considerations: a survey","volume":"55","author":"Jones","year":"2023","journal-title":"ACM Comput. Surv."},{"key":"10.1016\/j.patcog.2026.113365_bib0018","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110511","article-title":"Memory-Adaptive vision-and-Language navigation","volume":"153","author":"He","year":"2024","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113365_bib0019","series-title":"Findings of the Association for Computational Linguistics: ACL 2023, Toronto, Canada, July 9\u201314, 2023","first-page":"3043","article-title":"Aerial vision-and-dialog navigation","author":"Fan","year":"2023"},{"key":"10.1016\/j.patcog.2026.113365_bib0020","series-title":"Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing","article-title":"CityEQA: a hierarchical LLM agent on embodied question answering benchmark in city space","author":"Zhao","year":"2025"},{"key":"10.1016\/j.patcog.2026.113365_bib0021","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2025.103158","article-title":"UAVs Meet LLMs: overviews and perspectives towards agentic low-altitude mobility","volume":"122","author":"Tian","year":"2025","journal-title":"Inf. Fusion"},{"key":"10.1016\/j.patcog.2026.113365_bib0022","unstructured":"Y. Gao, Z. Wang, B. Zhao, Aerial vision-and-language navigation via semantic-topo-metric representation guided LLM reasoning, (2024). arXiv: 2410.08500."},{"key":"10.1016\/j.patcog.2026.113365_bib0023","article-title":"NavAgent: multi-scale urban street view fusion for UAV embodied vision-and-language navigation","author":"Liu","year":"2024","journal-title":"CoRR"},{"key":"10.1016\/j.patcog.2026.113365_bib0024","series-title":"International Conference on Representation Learning","first-page":"7292","article-title":"Towards realistic UAV vision-Language navigation: platform, benchmark, and methodology","volume":"2025","author":"Wang","year":"2025"},{"key":"10.1016\/j.patcog.2026.113365_bib0025","article-title":"AeroVerse: UAV-Agent benchmark suite for simulating, pre-training, finetuning, and evaluating aerospace embodied world models","author":"Yao","year":"2024","journal-title":"CoRR"},{"key":"10.1016\/j.patcog.2026.113365_bib0026","series-title":"2025IEEE International Conference on Robotics and Automation (ICRA)","first-page":"9490","article-title":"Spatialbot: precise spatial understanding with vision language models","author":"Cai","year":"2025"},{"key":"10.1016\/j.patcog.2026.113365_bib0027","series-title":"Proceedings of the 31st International Conference on Computational Linguistics, COLING 2025, Abu Dhabi, UAE, January 19\u201324, 2025","first-page":"2886","article-title":"Scaffolding coordinates to promote vision-language coordination in large multi-modal models","author":"Lei","year":"2025"},{"key":"10.1016\/j.patcog.2026.113365_bib0028","article-title":"Set-of-mark prompting unleashes extraordinary visual grounding in GPT-4V","volume":"abs\/2310.11441","author":"Yang","year":"2023","journal-title":"CoRR"},{"key":"10.1016\/j.patcog.2026.113365_bib0029","series-title":"Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","first-page":"32400","article-title":"UrbanVideo-bench: benchmarking vision-language models on embodied intelligence with video data in urban spaces","author":"Zhao","year":"2025"},{"key":"10.1016\/j.patcog.2026.113365_bib0030","series-title":"Proceedings of the Computer Vision and Pattern Recognition Conference","first-page":"10632","article-title":"Thinking in space: how multimodal large language models see, remember, and recall spaces","author":"Yang","year":"2025"},{"key":"10.1016\/j.patcog.2026.113365_bib0031","series-title":"Proceedings of the 33rd ACM International Conference on Multimedia","first-page":"12784","article-title":"Open3D-VQA: a benchmark for embodied spatial concept reasoning with multimodal large language model in open space","author":"Zhang","year":"2025"},{"key":"10.1016\/j.patcog.2026.113365_bib0032","series-title":"Proceedings of the 42nd International Conference on Machine Learning","first-page":"36340","article-title":"Imagine while reasoning in space: multimodal visualization-of-Thought","volume":"267","author":"Li","year":"2025"},{"key":"10.1016\/j.patcog.2026.113365_bib0033","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110511","article-title":"Memory-adaptive vision-and-language navigation","volume":"153","author":"He","year":"2024","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113365_bib0034","article-title":"TopV-Nav: unlocking the top-view spatial reasoning potential of MLLM for zero-shot object navigation","volume":"abs\/2411.16425","author":"Zhong","year":"2024","journal-title":"CoRR"},{"key":"10.1016\/j.patcog.2026.113365_bib0035","series-title":"Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics","first-page":"13032","article-title":"MapNav: a novel memory representation via annotated semantic maps for VLM-based vision-and-language navigation","author":"Zhang","year":"2025"},{"key":"10.1016\/j.patcog.2026.113365_bib0036","series-title":"International Conference on Machine Learning","first-page":"2778","article-title":"Curiosity-driven exploration by self-supervised prediction","author":"Pathak","year":"2017"},{"key":"10.1016\/j.patcog.2026.113365_bib0037","first-page":"531","article-title":"The parallelism tradeoff: limitations of log-precision transformers","volume":"11","author":"Merrill","year":"2023","journal-title":"Trans. Assoc. Comput. Ling."},{"key":"10.1016\/j.patcog.2026.113365_bib0038","series-title":"European Conference on Computer Vision","first-page":"38","article-title":"Grounding dino: marrying dino with grounded pre-training for open-set object detection","author":"Liu","year":"2024"},{"key":"10.1016\/j.patcog.2026.113365_bib0039","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"4015","article-title":"Segment anything","author":"Kirillov","year":"2023"},{"issue":"2","key":"10.1016\/j.patcog.2026.113365_bib0040","doi-asserted-by":"crossref","first-page":"316","DOI":"10.1007\/s11263-021-01554-9","article-title":"SensatUrban: learning semantics from urban-scale photogrammetric point clouds","volume":"130","author":"Hu","year":"2022","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.patcog.2026.113365_bib0041","series-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems","first-page":"77758","article-title":"CityRefer: geography-aware 3D visual grounding dataset on city-scale point cloud data","author":"Miyanishi","year":"2023"},{"key":"10.1016\/j.patcog.2026.113365_bib0042","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"6629","article-title":"Reinforced cross-modal matching and self-supervised imitation learning for vision-language navigation","author":"Wang","year":"2019"},{"key":"10.1016\/j.patcog.2026.113365_bib0043","series-title":"European Conference on Computer Vision","first-page":"104","article-title":"Beyond the nav-graph: vision-and-language navigation in continuous environments","author":"Krantz","year":"2020"},{"key":"10.1016\/j.patcog.2026.113365_bib0044","unstructured":"A. Hurst, A. Lerer, A. Radford, et al., Gpt-4o system card, (2024). arXiv: 2410.21276."},{"key":"10.1016\/j.patcog.2026.113365_bib0045","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"7641","article-title":"NavGPT: explicit reasoning in vision-and-language navigation with large language models","volume":"38","author":"Zhou","year":"2024"},{"key":"10.1016\/j.patcog.2026.113365_bib0046","unstructured":"S. Bai, K. Chen, J. Tang, et al., Qwen2. 5-VL Technical Report, (2025). arXiv: 2502.13923."}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326003304?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326003304?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T16:44:01Z","timestamp":1777567441000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0031320326003304"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":46,"alternative-id":["S0031320326003304"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113365","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"GeoNav: Empowering MLLMs with dual-scale geospatial reasoning for language-goal aerial navigation","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113365","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"113365"}}