{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,20]],"date-time":"2026-05-20T15:42:21Z","timestamp":1779291741777,"version":"3.51.4"},"reference-count":30,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,11,18]],"date-time":"2025-11-18T00:00:00Z","timestamp":1763424000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,11,18]],"date-time":"2025-11-18T00:00:00Z","timestamp":1763424000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,11,18]]},"DOI":"10.1109\/itsc60802.2025.11423269","type":"proceedings-article","created":{"date-parts":[[2026,3,16]],"date-time":"2026-03-16T20:10:23Z","timestamp":1773691823000},"page":"4437-4442","source":"Crossref","is-referenced-by-count":5,"title":["LogisticsVLN: Vision-Language Navigation for Low-Altitude Terminal Delivery Based on Agentic UAVs"],"prefix":"10.1109","author":[{"given":"Xinyuan","family":"Zhang","sequence":"first","affiliation":[{"name":"School of Artificial Intelligence, University of Chinese Academy of Sciences,Beijing,China,100049"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yonglin","family":"Tian","sequence":"additional","affiliation":[{"name":"Institute of Automation, Chinese Academy of Sciences,State Key Laboratory of Multimodal Artificial Intelligence Systems,Beijing,China,100190"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fei","family":"Lin","sequence":"additional","affiliation":[{"name":"Macau University of Science and Technology,Faculty of Innovation Engineering,Department of Engineering Science,Macau,China,999078"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yue","family":"Liu","sequence":"additional","affiliation":[{"name":"Institute of Automation, Chinese Academy of Sciences,State Key Laboratory of Multimodal Artificial Intelligence Systems,Beijing,China,100190"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jing","family":"Ma","sequence":"additional","affiliation":[{"name":"China Ship Research and Development Academy,Beijing,China,100101"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiao","family":"Wang","sequence":"additional","affiliation":[{"name":"School of Artificial Intelligence, Anhui University,Anhui,China,230601"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Korn\u00e9lia S\u00e1ra","family":"Szatm\u00e1ry","sequence":"additional","affiliation":[{"name":"Obuda University,Hungary"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fei-Yue","family":"Wang","sequence":"additional","affiliation":[{"name":"Chinese Academy of Sciences,State Key Laboratory for Management and Control of Complex Systems,Beijing,100190"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/JRFID.2024.3392943"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.4102\/jtscm.v12i0.336"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2025.103158"},{"key":"ref4","author":"Shah","year":"2023","journal-title":"ViNT: A Foundation Model for Visual Navigation"},{"key":"ref5","author":"Gao","year":"2025","journal-title":"OpenFly: A Versatile Toolchain and Largescale Benchmark for Aerial Vision-Language Navigation"},{"key":"ref6","author":"Wang","year":"2024","journal-title":"Towards Realistic UAV VisionLanguage Navigation: Platform, Benchmark, and Methodology"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/HRI61500.2025.10974117"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA55743.2025.11127476"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/3DV.2017.00081"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-67361-5_40"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr.2018.00387"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01000"},{"key":"ref13","author":"Shah","year":"2022","journal-title":"LM-Nav: Robotic Navigation with Large Pre-Trained Models of Language, Vision, and Action"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01411"},{"key":"ref15","author":"Limberg","year":"2024","journal-title":"Leveraging YOLO-World and GPT-4V LMMs for Zero-Shot Person Detection and Action Recognition in Drone Imagery"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/SMC58881.2025.11342598"},{"key":"ref17","author":"Zhao","year":"2023","journal-title":"Agent as Cerebrum, Controller as Cerebellum: Implementing an Embodied LMM-based Agent on Drones"},{"key":"ref18","author":"Liu","year":"2024","journal-title":"NavAgent: Multi-scale Urban Street View Fusion For UAV Embodied Vision-and-Language Navigation"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/MITS.2022.3159484"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/MITS.2024.3396430"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/SOLI63266.2024.10956119"},{"key":"ref22","year":"2025","journal-title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning"},{"key":"ref23","author":"Wei","year":"2022","journal-title":"Chain-of-Thought Prompting Elicits Reasoning in Large Language Models"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/IROS60139.2025.11246684"},{"key":"ref25","author":"Dosovitskiy","year":"2017","journal-title":"CARLA: An Open Urban Driving Simulator"},{"key":"ref26","year":"2024","journal-title":"GPT-4o System Card"},{"key":"ref27","author":"Anderson","year":"2018","journal-title":"On Evaluation of Embodied Navigation Agents"},{"key":"ref28","author":"Wang","year":"2024","journal-title":"Qwen2-VL: Enhancing Vision-Language Model\u2019s Perception of the World at Any Resolution"},{"key":"ref29","author":"Grattafiori","year":"2024","journal-title":"The Llama 3 Herd of Models"},{"key":"ref30","author":"Young","year":"2024","journal-title":"Yi: Open Foundation Models by 01.AI"}],"event":{"name":"2025 IEEE 28th International Conference on Intelligent Transportation Systems (ITSC)","location":"Gold Coast, Australia","start":{"date-parts":[[2025,11,18]]},"end":{"date-parts":[[2025,11,21]]}},"container-title":["2025 IEEE 28th International Conference on Intelligent Transportation Systems (ITSC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11422813\/11423000\/11423269.pdf?arnumber=11423269","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,17]],"date-time":"2026-03-17T05:48:08Z","timestamp":1773726488000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11423269\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,18]]},"references-count":30,"URL":"https:\/\/doi.org\/10.1109\/itsc60802.2025.11423269","relation":{},"subject":[],"published":{"date-parts":[[2025,11,18]]}}}