{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T15:59:46Z","timestamp":1784822386836,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":46,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62406092"],"award-info":[{"award-number":["62406092"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Shenzhen Science and Technology Program","award":["KJZD20240903100017022"],"award-info":[{"award-number":["KJZD20240903100017022"]}]},{"name":"Guangdong Basic and Applied Basic Research Foundation","award":["2025A1515010169"],"award-info":[{"award-number":["2025A1515010169"]}]},{"name":"National Natural Science Foundation of China Joint Funds","award":["U24B20175"],"award-info":[{"award-number":["U24B20175"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3758251","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:37:21Z","timestamp":1761377841000},"page":"13023-13029","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":6,"title":["UAV-ON: A Benchmark for Open-World Object Goal Navigation with Aerial Agents"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5054-0318","authenticated-orcid":false,"given":"Jianqiang","family":"Xiao","sequence":"first","affiliation":[{"name":"Harbin Institute of Technology, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-0692-7470","authenticated-orcid":false,"given":"Yuexuan","family":"Sun","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-1546-7801","authenticated-orcid":false,"given":"Yixin","family":"Shao","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-1134-1575","authenticated-orcid":false,"given":"Boxi","family":"Gan","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3731-6650","authenticated-orcid":false,"given":"Rongqiang","family":"Liu","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-3997-5524","authenticated-orcid":false,"given":"Yanjin","family":"Wu","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5658-5509","authenticated-orcid":false,"given":"Weili","family":"Guan","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3963-509X","authenticated-orcid":false,"given":"Xiang","family":"Deng","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11831-022-09742-7"},{"key":"e_1_3_2_1_2_1","volume-title":"Alexey Dosovitskiy, Saurabh Gupta, Vladlen Koltun, Jana Kosecka, Jitendra Malik, Roozbeh Mottaghi, Manolis Savva, et al.","author":"Anderson Peter","year":"2018","unstructured":"Peter Anderson, Angel Chang, Devendra Singh Chaplot, Alexey Dosovitskiy, Saurabh Gupta, Vladlen Koltun, Jana Kosecka, Jitendra Malik, Roozbeh Mottaghi, Manolis Savva, et al. 2018. On evaluation of embodied navigation agents. arXiv preprint arXiv:1807.06757 (2018)."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.petrol.2021.109633"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ijtst.2017.02.001"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.paerosci.2020.100617"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01445"},{"key":"e_1_3_2_1_7_1","first-page":"4247","article-title":"Object goal navigation using goal-oriented semantic exploration","volume":"33","author":"Chaplot Devendra Singh","year":"2020","unstructured":"Devendra Singh Chaplot, Dhiraj Prakashchand Gandhi, Abhinav Gupta, and Ruslan Salakhutdinov. 2020. Object goal navigation using goal-oriented semantic exploration. Advances in Neural Information Processing Systems 33 (2020), 4247--4258.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_8_1","volume-title":"Zero-shot object searching using large-scale object relationship prior. arXiv preprint arXiv:2303.06228","author":"Chen Hongyi","year":"2023","unstructured":"Hongyi Chen, Ruinian Xu, Shuo Cheng, Patricio A. Vela, and Danfei Xu. 2023. Zero-shot object searching using large-scale object relationship prior. arXiv preprint arXiv:2303.06228 (2023)."},{"key":"e_1_3_2_1_9_1","volume-title":"European Conference on Computer Vision (ECCV). Springer Nature Switzerland, 213--231","author":"Chu Meng","year":"2024","unstructured":"Meng Chu, Zhedong Zheng, Wei Ji, Tingyu Wang, and Tat-Seng Chua. 2024. Towards Natural Language-Guided Drones: GeoText-1652 Benchmark with Spatial Relation Matching. In European Conference on Computer Vision (ECCV). Springer Nature Switzerland, 213--231."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00323"},{"key":"e_1_3_2_1_11_1","unstructured":"Epic Games. [n.d.]. Unreal Engine. https:\/\/www.unrealengine.com"},{"key":"e_1_3_2_1_12_1","unstructured":"Yunpeng Gao Chenhui Li Zhongrui You Junli Liu et al. 2025. OpenFly: A Versatile Toolchain and Large-scale Benchmark for Aerial Vision-Language Navigation. arXiv preprint arXiv:2502.18041 (2025)."},{"key":"e_1_3_2_1_13_1","unstructured":"Yunpeng Gao ZhigangWang Pengfei Han Linglin Jing et al. 2024. Aerial Visionand-Language Navigation via Semantic-Topo-Metric Representation Guided LLM Reasoning. arXiv preprint arXiv:2410.08500 (2024)."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1007\/978--3-030--75067--1_19"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/tssc.1968"},{"key":"e_1_3_2_1_16_1","volume-title":"General Scene Adaptation for Vision-and-Language Navigation. arXiv preprint arXiv:2501.17403","author":"Hong Haodong","year":"2025","unstructured":"Haodong Hong, Yanyuan Qiao, Sen Wang, Jiajun Liu, and Qi Wu. 2025. General Scene Adaptation for Vision-and-Language Navigation. arXiv preprint arXiv:2501.17403 (2025)."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3570723"},{"key":"e_1_3_2_1_18_1","unstructured":"Eric Kolve Roozbeh Mottaghi Winson Han et al. 2017. Ai2-thor: An interactive 3d environment for visual ai. arXiv preprint arXiv:1712.05474 (2017)."},{"key":"e_1_3_2_1_19_1","unstructured":"Jungdae Lee TaikiMiyanishi Shuhei Kurita Koya Sakamoto et al. 2024. CityNav: Language-Goal Aerial Navigation Dataset with Geographic Information. arXiv preprint arXiv:2406.14240 (2024)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2022.3189214"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01411"},{"key":"e_1_3_2_1_22_1","unstructured":"Youzhi Liu Fanglong Yao Yuanchang Yue Guangluan Xu et al. 2024. NavAgent: Multi-scale Urban Street View Fusion For UAV Embodied Vision-and-Language Navigation. arXiv preprint arXiv:2411.08579 (2024)."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1080\/10095020.2017.1420509"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/IROS47612.2022.9981646"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.techfore.2020.119293"},{"key":"e_1_3_2_1_26_1","unstructured":"OpenAI Josh Achiam Steven Adler Sandhini Agarwal et al. 2023. Gpt-4 technical report. arXiv preprint arXiv:2303.08774 (2023)."},{"key":"e_1_3_2_1_27_1","volume-title":"Yibing Xie, Alessandro Gardi, and Roberto Sabatini.","author":"Pongsakornsathien Nichakorn","year":"2025","unstructured":"Nichakorn Pongsakornsathien, Nour El-Din Safwat, Yibing Xie, Alessandro Gardi, and Roberto Sabatini. 2025. Advances in low-altitude airspace management for uncrewed aircraft and advanced air mobility. Progress in Aerospace Sciences (2025), 101085."},{"key":"e_1_3_2_1_28_1","unstructured":"Alec Radford JongWook Kim Chris Hallacy Aditya Ramesh et al. 2021. Learning transferable visual models from natural language supervision. PmLR 8748--8763."},{"key":"e_1_3_2_1_29_1","unstructured":"Santhosh K. Ramakrishnan Aaron Gokaslan Erik Wijmans Oleksandr Maksymets et al. 2021. Habitat-Matterport 3D dataset (HM3D): 1000 large-scale 3D environments for embodied AI. arXiv preprint arXiv:2109.08238 (2021)."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01716"},{"key":"e_1_3_2_1_31_1","volume-title":"Learning from demonstration. Advances in neural information processing systems 9","author":"Schaal Stefan","year":"1996","unstructured":"Stefan Schaal. 1996. Learning from demonstration. Advances in neural information processing systems 9 (1996)."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/2750675.2750683"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-67361-5_40"},{"key":"e_1_3_2_1_34_1","volume-title":"Barto","author":"Sutton Richard S.","year":"1998","unstructured":"Richard S. Sutton and Andrew G. Barto. 1998. Reinforcement learning: An introduction. Vol. 1. MIT press Cambridge."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.compag.2020.105523"},{"key":"e_1_3_2_1_36_1","unstructured":"Peng Wang Shuai Bai Sinan Tan Shijie Wang et al. 2024. Qwen2-vl: Enhancing vision-language model's perception of the world at any resolution. arXiv preprint arXiv:2409.12191 (2024)."},{"key":"e_1_3_2_1_37_1","unstructured":"XiangyuWang Donglin Yang ZiqinWang Hohin Kwan et al. 2024. Towards Realistic UAV Vision-Language Navigation: Platform Benchmark and Methodology. arXiv preprint arXiv:2410.07087 (2024)."},{"key":"e_1_3_2_1_38_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 9068--9079","author":"Xia Fei","year":"2018","unstructured":"Fei Xia, Amir R. Zamir, Zhiyang He, Alexander Sax, et al. 2018. Gibson env: Realworld perception for embodied agents. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 9068--9079."},{"key":"e_1_3_2_1_39_1","unstructured":"Haotian Xu Yue Hu Chen Gao Zhengqiu Zhu et al. 2025. GeoNav: Empowering MLLMs with Explicit Geospatial Reasoning Abilities for Language-Goal Aerial Navigation. arXiv preprint arXiv:2504.09587 (2025)."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01581"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2021.3068906"},{"key":"e_1_3_2_1_42_1","volume-title":"Safe-vln: Collision avoidance for vision-and-language navigation of autonomous robots operating in continuous environments","author":"Yue Lu","year":"2024","unstructured":"Lu Yue, Dongliang Zhou, Liang Xie, Feitian Zhang, et al. 2024. Safe-vln: Collision avoidance for vision-and-language navigation of autonomous robots operating in continuous environments. IEEE Robotics and Automation Letters (2024)."},{"key":"e_1_3_2_1_43_1","volume-title":"Aerial Vision-and-Language Navigation with Grid-based View Selection and Map Construction. arXiv preprint arXiv:2503.11091","author":"Zhao Ganlong","year":"2025","unstructured":"Ganlong Zhao, Guanbin Li, Jia Pan, and Yizhou Yu. 2025. Aerial Vision-and-Language Navigation with Grid-based View Selection and Map Construction. arXiv preprint arXiv:2503.11091 (2025)."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2023.3320014"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"crossref","unstructured":"Qianfan Zhao Lu Zhang Bin He Hong Qiao and Zhiyong Liu. 2022. Zero-shot object goal visual navigation. arXiv preprint arXiv:2206.07423.","DOI":"10.1109\/ICRA48891.2023.10161289"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ress"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3758251","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:59:26Z","timestamp":1765342766000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3758251"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":46,"alternative-id":["10.1145\/3746027.3758251","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3758251","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}