{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T12:23:47Z","timestamp":1783081427884,"version":"3.54.6"},"reference-count":48,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100018606","name":"Natural Products Research Center of Guizhou Province","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100018606","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100010828","name":"Guizhou Province Department of Education","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100010828","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003459","name":"Guizhou University","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100003459","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["12205062"],"award-info":[{"award-number":["12205062"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100005329","name":"Natural Science Foundation of Guizhou Province","doi-asserted-by":"publisher","award":["LH[2017]7226"],"award-info":[{"award-number":["LH[2017]7226"]}],"id":[{"id":"10.13039\/501100005329","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100005329","name":"Natural Science Foundation of Guizhou Province","doi-asserted-by":"publisher","award":["[2016]1054"],"award-info":[{"award-number":["[2016]1054"]}],"id":[{"id":"10.13039\/501100005329","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Knowledge-Based Systems"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1016\/j.knosys.2026.116308","type":"journal-article","created":{"date-parts":[[2026,5,27]],"date-time":"2026-05-27T15:57:49Z","timestamp":1779897469000},"page":"116308","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Semantic-aware and spatially-grounded multi-expert network for continuous vision-and-language navigation"],"prefix":"10.1016","volume":"347","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-2756-323X","authenticated-orcid":false,"given":"Ronghu","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9191-2790","authenticated-orcid":false,"given":"Ziyan","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-8185-4190","authenticated-orcid":false,"given":"Weidong","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-2423-1564","authenticated-orcid":false,"given":"Banghai","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-8280-2286","authenticated-orcid":false,"given":"Ziyou","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.knosys.2026.116308_b1","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"3674","article-title":"Vision-and-language navigation: Interpreting visually grounded navigation instructions in real environments","author":"Anderson","year":"2018"},{"key":"10.1016\/j.knosys.2026.116308_b2","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"6629","article-title":"Reinforced cross-modal matching and self-supervised imitation learning for vision-language navigation","author":"Wang","year":"2019"},{"key":"10.1016\/j.knosys.2026.116308_b3","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"8455","article-title":"Structured scene memory for vision-language navigation","author":"Wang","year":"2021"},{"key":"10.1016\/j.knosys.2026.116308_b4","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"1643","article-title":"VLN BERT: A recurrent vision-and-language BERT for navigation","author":"Hong","year":"2021"},{"key":"10.1016\/j.knosys.2026.116308_b5","series-title":"Advances in Neural Information Processing Systems","first-page":"5834","article-title":"History aware multi-modal transformer for vision-and-language navigation","author":"Chen","year":"2021"},{"key":"10.1016\/j.knosys.2026.116308_b6","series-title":"Computer Vision \u2013 ECCV 2020: 16th European Conference","first-page":"104","article-title":"Beyond the nav-graph: Vision-and-language navigation in continuous environments","author":"Krantz","year":"2020"},{"key":"10.1016\/j.knosys.2026.116308_b7","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"15439","article-title":"Bridging the gap between learning in discrete and continuous environments for vision-and-language navigation","author":"Hong","year":"2022"},{"key":"10.1016\/j.knosys.2026.116308_b8","series-title":"2021 IEEE\/CVF International Conference on Computer Vision","first-page":"15162","article-title":"Waypoint models for instruction-guided navigation in continuous environments","author":"Krantz","year":"2021"},{"key":"10.1016\/j.knosys.2026.116308_b9","series-title":"Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing","first-page":"4018","article-title":"Language-aligned waypoint (law) supervision for vision-and-language navigation in continuous environments","author":"Raychaudhuri","year":"2021"},{"issue":"7","key":"10.1016\/j.knosys.2026.116308_b10","doi-asserted-by":"crossref","first-page":"5130","DOI":"10.1109\/TPAMI.2024.3386695","article-title":"Etpnav: Evolving topological planning for vision-language navigation in continuous environments","volume":"47","author":"An","year":"2024","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.knosys.2026.116308_b11","series-title":"IEEE International Conference on Robotics and Automation","first-page":"6710","article-title":"Open-nav: Exploring zero-shot vision-and-language navigation in continuous environment with open-source llms","author":"Qiao","year":"2025"},{"key":"10.1016\/j.knosys.2026.116308_b12","series-title":"IEEE\/RSJ International Conference on Intelligent Robots and Systems","article-title":"SmartWay: Enhanced waypoint prediction and backtracking for zero-shot vision-and-language navigation","author":"Shi","year":"2025"},{"key":"10.1016\/j.knosys.2026.116308_b13","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"13137","article-title":"Towards learning a generic agent for vision-and-language navigation via pre-training","author":"Hao","year":"2020"},{"key":"10.1016\/j.knosys.2026.116308_b14","series-title":"Advances in Neural Information Processing Systems","first-page":"11276","article-title":"Topological planning with transformers for vision-and-language navigation","author":"Chen","year":"2021"},{"key":"10.1016\/j.knosys.2026.116308_b15","series-title":"Computer Vision \u2013 ECCV 2022: 18th European Conference","first-page":"588","article-title":"Sim-2-sim transfer for vision-and-language navigation in continuous environments","author":"Krantz","year":"2022"},{"key":"10.1016\/j.knosys.2026.116308_b16","series-title":"Advances in Neural Information Processing Systems","article-title":"Evolving graphical planner: Contextual global planning for vision-and-language navigation","author":"Deng","year":"2020"},{"key":"10.1016\/j.knosys.2026.116308_b17","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"16537","article-title":"Think global, act local: Dual-scale graph transformer for vision-and-language navigation","author":"Chen","year":"2022"},{"key":"10.1016\/j.knosys.2026.116308_b18","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"23212","article-title":"Geovln: Learning geometry-enhanced visual representation with slot attention for vision-and-language navigation","author":"Huo","year":"2023"},{"key":"10.1016\/j.knosys.2026.116308_b19","series-title":"Proceedings of the Thirty-Second International Joint Conference on Artificial Intelligence","first-page":"1479","article-title":"A dual semantic-aware recurrent global-adaptive network for vision-and-language navigation","author":"Wang","year":"2023"},{"key":"10.1016\/j.knosys.2026.116308_b20","series-title":"IEEE\/RSJ International Conference on Intelligent Robots and Systems","first-page":"7726","article-title":"Enhanced language-guided robot navigation with panoramic semantic depth perception and cross-modal fusion","author":"Wang","year":"2024"},{"key":"10.1016\/j.knosys.2026.116308_b21","series-title":"Advances in Neural Information Processing Systems","article-title":"Hierarchical semantic-augmented navigation: Optimal transport and graph-driven reasoning for vision-language navigation","author":"Fang","year":"2025"},{"key":"10.1016\/j.knosys.2026.116308_b22","doi-asserted-by":"crossref","DOI":"10.1016\/j.neunet.2025.107320","article-title":"Panogen++: Domain-adapted text-guided panoramic environment generation for vision-and-language navigation","volume":"187","author":"Wang","year":"2025","journal-title":"Neural Netw."},{"issue":"12","key":"10.1016\/j.knosys.2026.116308_b23","doi-asserted-by":"crossref","first-page":"8534","DOI":"10.1109\/TPAMI.2024.3407759","article-title":"Correctable landmark discovery via large models for vision-language navigation","volume":"46","author":"Lin","year":"2024","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.knosys.2026.116308_b24","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2024.112735","article-title":"Adaptive cross-modal experts network with uncertainty-driven fusion for vision-language navigation","volume":"307","author":"Wu","year":"2025","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.knosys.2026.116308_b25","series-title":"Computer Vision \u2013 ECCV 2024: 18th European Conference","first-page":"459","article-title":"LLM as copilot for coarse-grained vision-and-language navigation","author":"Qiao","year":"2024"},{"key":"10.1016\/j.knosys.2026.116308_b26","series-title":"2023 IEEE\/CVF International Conference on Computer Vision","first-page":"15625","article-title":"Gridmm: Grid memory map for vision-and-language navigation","author":"Wang","year":"2023"},{"issue":"6","key":"10.1016\/j.knosys.2026.116308_b27","doi-asserted-by":"crossref","first-page":"15460","DOI":"10.1109\/LRA.2024.3387171","article-title":"Safe-vln: Collision avoidance for vision-and-language navigation of autonomous robots operating in continuous environments","volume":"9","author":"Yue","year":"2024","journal-title":"IEEE Robot. Autom. Lett."},{"key":"10.1016\/j.knosys.2026.116308_b28","series-title":"Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics","first-page":"13032","article-title":"Mapnav: A novel memory representation via annotated semantic maps for vlm-based vision-and-language navigation","author":"Zhang","year":"2025"},{"key":"10.1016\/j.knosys.2026.116308_b29","series-title":"IEEE International Conference on Robotics and Automation","first-page":"5236","article-title":"VLN-KHVR: Knowledge-and-history aware visual representation for continuous vision-and-language navigation","author":"Kong","year":"2025"},{"key":"10.1016\/j.knosys.2026.116308_b30","series-title":"Robotics: Science and Systems","article-title":"NavID: Video-based VLM plans the next step for vision-and-language navigation","author":"Zhang","year":"2024"},{"key":"10.1016\/j.knosys.2026.116308_b31","series-title":"2023 IEEE\/CVF International Conference on Computer Vision","first-page":"15758","article-title":"March in chat: Interactive prompting for remote embodied referring expression","author":"Qiao","year":"2023"},{"key":"10.1016\/j.knosys.2026.116308_b32","series-title":"38th AAAI Conference on Artificial Intelligence","first-page":"18924","article-title":"VELMA: Verbalization embodiment of LLM agents for vision and language navigation in street view","author":"Schumann","year":"2024"},{"key":"10.1016\/j.knosys.2026.116308_b33","series-title":"Advances in Neural Information Processing Systems","article-title":"Find what you want: Learning demand-conditioned object attribute space for demand-driven navigation","author":"Wang","year":"2024"},{"key":"10.1016\/j.knosys.2026.116308_b34","series-title":"38th AAAI Conference on Artificial Intelligence","first-page":"7641","article-title":"NavGPT: Explicit reasoning in vision-and-language navigation with large language models","author":"Zhou","year":"2024"},{"key":"10.1016\/j.knosys.2026.116308_b35","series-title":"Proceedings of the 2024 Conference on Robot Learning","article-title":"Instructnav: Zero-shot system for generic instruction navigation in unexplored environment","author":"Long","year":"2024"},{"key":"10.1016\/j.knosys.2026.116308_b36","series-title":"IEEE International Conference on Robotics and Automation","first-page":"17380","article-title":"Discuss before moving: Visual language navigation via multi-expert discussions","author":"Long","year":"2024"},{"issue":"7","key":"10.1016\/j.knosys.2026.116308_b37","doi-asserted-by":"crossref","first-page":"5945","DOI":"10.1109\/TPAMI.2025.3554559","article-title":"Navcot: Boosting llm-based vision-and-language navigation via learning disentangled reasoning","volume":"47","author":"Lin","year":"2025","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.knosys.2026.116308_b38","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"3783","article-title":"General object foundation model for images and videos at scale","author":"Wu","year":"2024"},{"key":"10.1016\/j.knosys.2026.116308_b39","series-title":"37th AAAI Conference on Artificial Intelligence","first-page":"1386","article-title":"Layout-aware dreamer for embodied visual referring expression grounding","author":"Li","year":"2023"},{"key":"10.1016\/j.knosys.2026.116308_b40","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"10803","article-title":"Improving vision-and-language navigation by generating future-view image semantics","author":"Li","year":"2023"},{"key":"10.1016\/j.knosys.2026.116308_b41","series-title":"2019 IEEE\/CVF International Conference on Computer Vision","first-page":"9339","article-title":"Habitat: A platform for embodied ai research","author":"Savva","year":"2019"},{"key":"10.1016\/j.knosys.2026.116308_b42","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"9982","article-title":"Reverie: remote embodied visual referring expression in real indoor environments","author":"Qi","year":"2020"},{"key":"10.1016\/j.knosys.2026.116308_b43","series-title":"International Conference on Learning Representations","article-title":"Dd-ppo: Learning near-perfect pointgoal navigators from 2.5 billion frames","author":"Wijmans","year":"2019"},{"key":"10.1016\/j.knosys.2026.116308_b44","series-title":"Proceedings of EMNLP-IJCNLP","first-page":"5099","article-title":"LXMERT: Learning cross-modality encoder representations from transformers","author":"Li","year":"2019"},{"key":"10.1016\/j.knosys.2026.116308_b45","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"15460","article-title":"Cross-modal map learning for vision and language navigation","author":"Georgakis","year":"2022"},{"key":"10.1016\/j.knosys.2026.116308_b46","series-title":"2023 IEEE\/CVF International Conference on Computer Vision","article-title":"BEV-BERT: Multi-modal map pre-training for language-guided navigation","author":"An","year":"2023"},{"key":"10.1016\/j.knosys.2026.116308_b47","doi-asserted-by":"crossref","first-page":"6690","DOI":"10.1109\/TMM.2025.3586105","article-title":"MossVLN: Memory-observation synergistic system for continuous vision-language navigation","author":"Yu","year":"2025","journal-title":"IEEE Trans. MultiMed."},{"key":"10.1016\/j.knosys.2026.116308_b48","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"13753","article-title":"Lookahead exploration with neural radiance representation for continuous vision-language navigation","author":"Wang","year":"2024"}],"container-title":["Knowledge-Based Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0950705126010348?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0950705126010348?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T11:54:37Z","timestamp":1783079677000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0950705126010348"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":48,"alternative-id":["S0950705126010348"],"URL":"https:\/\/doi.org\/10.1016\/j.knosys.2026.116308","relation":{},"ISSN":["0950-7051"],"issn-type":[{"value":"0950-7051","type":"print"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Semantic-aware and spatially-grounded multi-expert network for continuous vision-and-language navigation","name":"articletitle","label":"Article Title"},{"value":"Knowledge-Based Systems","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.knosys.2026.116308","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"116308"}}