{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,23]],"date-time":"2026-06-23T18:34:33Z","timestamp":1782239673600,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":38,"publisher":"ACM","funder":[{"name":"National Science Foundation","award":["2218063"],"award-info":[{"award-number":["2218063"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3728482.3757386","type":"proceedings-article","created":{"date-parts":[[2025,11,3]],"date-time":"2025-11-03T21:21:35Z","timestamp":1762204895000},"page":"21-25","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["VICI: VLM-Instructed Cross-view Image-localisation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-6344-9604","authenticated-orcid":false,"given":"Xiaohan","family":"Zhang","sequence":"first","affiliation":[{"name":"University of Vermont, Burlington, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3877-4736","authenticated-orcid":false,"given":"Tavis","family":"Shore","sequence":"additional","affiliation":[{"name":"University of Surrey, Guildford, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3957-7061","authenticated-orcid":false,"given":"Chen","family":"Chen","sequence":"additional","affiliation":[{"name":"University of Central Florida, Orlando, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4904-4349","authenticated-orcid":false,"given":"Oscar","family":"Mendez","sequence":"additional","affiliation":[{"name":"Locus Robotics, Boston, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8637-5054","authenticated-orcid":false,"given":"Simon","family":"Hadfield","sequence":"additional","affiliation":[{"name":"University of Surrey, Guildford, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5051-7719","authenticated-orcid":false,"given":"Safwan","family":"Wshah","sequence":"additional","affiliation":[{"name":"University of Vermont, Burlington, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,11,3]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Barkin Dagda Muhammad Awais and Saber Fallah. 2025. GeoVLM: Improving Automated Vehicle Geolocalisation Using Vision-Language Matching. arXiv:2505.13669 [cs.CV] https:\/\/arxiv.org\/abs\/2505.13669"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"crossref","unstructured":"Fabian Deuser Konrad Habel and Norbert Oswald. 2023. Sample4Geo: Hard Negative Sampling For Cross-View Geo-Localisation. arXiv:2303.11851 [cs.CV]","DOI":"10.1109\/ICCV51070.2023.01545"},{"key":"e_1_3_2_1_3_1","volume-title":"International Conference on Learning Representations.","author":"Dosovitskiy Alexey","year":"2021","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, Jakob Uszkoreit, and Neil Houlsby. 2021. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_4_1","unstructured":"Google Cloud. 2025. Gemini 2.5 Flash Model. https:\/\/cloud.google.com\/vertex-ai\/generative-ai\/docs\/models\/gemini\/2-5-flash. Accessed: 2025-06-15."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i3.27999"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSTARS.2024.3502160"},{"key":"e_1_3_2_1_7_1","volume-title":"Forty-first International Conference on Machine Learning.","author":"Li Ling","year":"2024","unstructured":"Ling Li, Yu Ye, Bingchuan Jiang, and Wei Zeng. 2024. Georeasoner: Geo-localization with reasoning in street views using a large vision-language model. In Forty-first International Conference on Machine Learning."},{"key":"e_1_3_2_1_8_1","volume-title":"Visual instruction tuning. Advances in neural information processing systems","author":"Liu Haotian","year":"2023","unstructured":"Haotian Liu, Chunyuan Li, Qingyang Wu, and Yong Jae Lee. 2023. Visual instruction tuning. Advances in neural information processing systems, Vol. 36 (2023), 34892-34916."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00577"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01167"},{"key":"e_1_3_2_1_11_1","volume-title":"Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101","author":"Loshchilov Ilya","year":"2017","unstructured":"Ilya Loshchilov and Frank Hutter. 2017. Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101 (2017)."},{"key":"e_1_3_2_1_12_1","volume-title":"European Conference on Computer Vision. Springer, 214-230","author":"Mi Li","year":"2024","unstructured":"Li Mi, Chang Xu, Javiera Castillo-Navarro, Syrielle Montariol, Wen Yang, Antoine Bosselut, and Devis Tuia. 2024. Congeo: Robust cross-view geo-localization across ground view variations. In European Conference on Computer Vision. Springer, 214-230."},{"key":"e_1_3_2_1_13_1","unstructured":"OpenAI. 2024. ChatGPT (July 5 Version). https:\/\/chat.openai.com\/. Accessed: 2025-07-05."},{"key":"e_1_3_2_1_14_1","unstructured":"Maxime Oquab Timoth\u00e9e Darcet Th\u00e9o Moutakanni Huy Vo Marc Szafraniec Vasil Khalidov Pierre Fernandez Daniel Haziza Francisco Massa Alaaeldin El-Nouby et al. 2023. Dinov2: Learning robust visual features without supervision. arXiv preprint arXiv:2304.07193 (2023)."},{"key":"e_1_3_2_1_15_1","volume-title":"Pytorch: An imperative style, high-performance deep learning library. Advances in neural information processing systems","author":"Paszke Adam","year":"2019","unstructured":"Adam Paszke, Sam Gross, Francisco Massa, Adam Lerer, James Bradbury, Gregory Chanan, Trevor Killeen, Zeming Lin, Natalia Gimelshein, Luca Antiga, et al., 2019. Pytorch: An imperative style, high-performance deep learning library. Advances in neural information processing systems, Vol. 32 (2019)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00412"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00412"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2025.3546513"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV61041.2025.00667"},{"key":"e_1_3_2_1_20_1","unstructured":"Gemini Team Rohan Anil Sebastian Borgeaud Jean-Baptiste Alayrac Jiahui Yu Radu Soricut Johan Schalkwyk Andrew M Dai Anja Hauth Katie Millican et al. 2023. Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:2312.11805 (2023)."},{"key":"e_1_3_2_1_21_1","volume-title":"Visual autoregressive modeling: Scalable image generation via next-scale prediction. Advances in neural information processing systems","author":"Tian Keyu","year":"2024","unstructured":"Keyu Tian, Yi Jiang, Zehuan Yuan, Bingyue Peng, and Liwei Wang. 2024. Visual autoregressive modeling: Scalable image generation via next-scale prediction. Advances in neural information processing systems, Vol. 37 (2024), 84839-84865."},{"key":"e_1_3_2_1_22_1","volume-title":"Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971","author":"Touvron Hugo","year":"2023","unstructured":"Hugo Touvron, Thibaut Lavril, Gautier Izacard, Xavier Martinet, Marie-Anne Lachaux, Timoth\u00e9e Lacroix, Baptiste Rozi\u00e8re, Naman Goyal, Eric Hambro, Faisal Azhar, et al., 2023. Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971 (2023)."},{"key":"e_1_3_2_1_23_1","volume-title":"Attention is all you need. Advances in neural information processing systems","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems, Vol. 30 (2017)."},{"key":"e_1_3_2_1_24_1","volume-title":"The 3rd Workshop on UAVs in Multimedia: Capturing the World from a New Perspective. In Proceedings of the 33rd ACM International Conference on Multimedia Workshop.","author":"Wang Tingyu","year":"2025","unstructured":"Tingyu Wang, Yujiao Shi, Fabian Deuser, Shaofei Huang, Guosheng Hu, Si Liu, Zhedong Zheng, and Roger Zimmermann. 2025a. The 3rd Workshop on UAVs in Multimedia: Capturing the World from a New Perspective. In Proceedings of the 33rd ACM International Conference on Multimedia Workshop."},{"key":"e_1_3_2_1_25_1","volume-title":"The 3rd Workshop on UAVs in Multimedia: Capturing the World from a New Perspective. In Proceedings of the 33rd ACM International Conference on Multimedia Workshop.","author":"Wang Tingyu","year":"2025","unstructured":"Tingyu Wang, Yujiao Shi, Fabian Deuser, Shaofei Huang, Guosheng Hu, Si Liu, Zhedong Zheng, and Roger Zimmermann. 2025b. The 3rd Workshop on UAVs in Multimedia: Capturing the World from a New Perspective. In Proceedings of the 33rd ACM International Conference on Multimedia Workshop."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2021.3061265"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-023-01942-3"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2015.7301385"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.451"},{"key":"e_1_3_2_1_30_1","volume-title":"Zhu","author":"Yang Hongji","year":"2021","unstructured":"Hongji Yang, Xiufan Lu, and Ying J. Zhu. 2021. Cross-view Geo-localization with Layer-to-Layer Transformer. In Neural Information Processing Systems."},{"key":"e_1_3_2_1_31_1","volume-title":"Where am I? Cross-View Geo-localization with Natural Language Descriptions. arXiv preprint arXiv:2412.17007","author":"Ye Junyan","year":"2024","unstructured":"Junyan Ye, Honglin Lin, Leyan Ou, Dairong Chen, Zihao Wang, Conghui He, and Weijia Li. 2024. Where am I? Cross-View Geo-localization with Natural Language Descriptions. arXiv preprint arXiv:2412.17007 (2024)."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2024.3443652"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i3.25457"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV56688.2023.00293"},{"key":"e_1_3_2_1_35_1","volume-title":"University-1652: A Multi-view Multi-source Benchmark for Drone-based Geo-localization. ACM Multimedia","author":"Zheng Zhedong","year":"2020","unstructured":"Zhedong Zheng, Yunchao Wei, and Yi Yang. 2020. University-1652: A Multi-view Multi-source Benchmark for Drone-based Geo-localization. ACM Multimedia (2020)."},{"key":"e_1_3_2_1_36_1","first-page":"1152","volume-title":"TransGeo: Transformer Is All You Need for Cross-view Image Geo-localization. 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Zhu Sijie","year":"2022","unstructured":"Sijie Zhu, Mubarak Shah, and Chen Chen. 2022. TransGeo: Transformer Is All You Need for Cross-view Image Geo-localization. 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2022), 1152-1161."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00364"},{"key":"e_1_3_2_1_38_1","unstructured":"Yingying Zhu Hongji Yang Yuxin Lu and Qiang Huang. 2023. Simple Effective and General: A New Backbone for Cross-view Image Geo-localization. arXiv:2302.01572 [cs.CV]"}],"event":{"name":"MM '25:The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 3rd International Workshop on UAVs in Multimedia: Capturing the World from a New Perspective"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3728482.3757386","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,3]],"date-time":"2025-11-03T21:22:16Z","timestamp":1762204936000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3728482.3757386"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":38,"alternative-id":["10.1145\/3728482.3757386","10.1145\/3728482"],"URL":"https:\/\/doi.org\/10.1145\/3728482.3757386","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-11-03","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}