{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,4]],"date-time":"2026-05-04T05:44:40Z","timestamp":1777873480869,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":50,"publisher":"ACM","funder":[{"name":"Research Impact Fund","award":["No.R1015-23"],"award-info":[{"award-number":["No.R1015-23"]}]},{"name":"Collaborative Research Fund","award":["No.C1043-24GF"],"award-info":[{"award-number":["No.C1043-24GF"]}]},{"name":"Huawei Innovation Research Program"},{"name":"Huawei Fellowship"},{"name":"CCF-Tencent Open Fund"},{"name":"Tencent Rhino-Bird Focused Research Program"},{"name":"CCF-Alimama Tech Kangaroo Fund","award":["No. 2024002"],"award-info":[{"award-number":["No. 2024002"]}]},{"name":"CCF-Ant Research Fund"},{"name":"Kuaishou"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,8,3]]},"DOI":"10.1145\/3711896.3737141","type":"proceedings-article","created":{"date-parts":[[2025,8,1]],"date-time":"2025-08-01T13:30:13Z","timestamp":1754055013000},"page":"814-825","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Swarm Intelligence in Geo-Localization: A Multi-Agent Large Vision-Language Model Collaborative Framework"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3478-964X","authenticated-orcid":false,"given":"Xiao","family":"Han","sequence":"first","affiliation":[{"name":"City University of Hong Kong, Hong Kong, China and Zhejiang University of Technology, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4817-482X","authenticated-orcid":false,"given":"Chen","family":"Zhu","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4570-643X","authenticated-orcid":false,"given":"Hengshu","family":"Zhu","sequence":"additional","affiliation":[{"name":"Computer Network Information Center, Chinese Academy of Sciences, Beijing, China and University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2926-4416","authenticated-orcid":false,"given":"Xiangyu","family":"Zhao","sequence":"additional","affiliation":[{"name":"City University of Hong Kong, Hong Kong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,8,3]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"NetVLAD: CNN Architecture for Weakly Supervised Place Recognition","author":"Arandjelovic Relja","year":"2018","unstructured":"Relja Arandjelovic, Petr Gron\u00e1t, Akihiko Torii, Tom\u00e1s Pajdla, and Josef Sivic. 2018. NetVLAD: CNN Architecture for Weakly Supervised Place Recognition. IEEE Trans. Pattern Anal. Mach. Intell.(2018), 1437-1451."},{"key":"e_1_3_2_2_2_1","first-page":"4868","article-title":"Rethinking Visual Geo-localization for Large-Scale Applications","author":"Berton Gabriele Moreno","year":"2022","unstructured":"Gabriele Moreno Berton, Carlo Masone, and Barbara Caputo. 2022. Rethinking Visual Geo-localization for Large-Scale Applications. In Proc. of CVPR. 4868-4878.","journal-title":"Proc. of CVPR."},{"key":"e_1_3_2_2_3_1","volume-title":"Yuanzhi Li, Scott M. Lundberg, Harsha Nori, Hamid Palangi, Marco T\u00falio Ribeiro, and Yi Zhang.","author":"Bubeck S\u00e9bastien","year":"2023","unstructured":"S\u00e9bastien Bubeck, Varun Chandrasekaran, Ronen Eldan, Johannes Gehrke, Eric Horvitz, Ece Kamar, Peter Lee, Yin Tat Lee, Yuanzhi Li, Scott M. Lundberg, Harsha Nori, Hamid Palangi, Marco T\u00falio Ribeiro, and Yi Zhang. 2023. Sparks of Artificial General Intelligence: Early experiments with GPT-4. CoRR(2023)."},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"crossref","unstructured":"Mark Campbell and Matt Wheeler. 2006. A vision based geolocation tracking system for uav's. In AIAA Guidance Navigation and Control Conference and Exhibit. 6246.","DOI":"10.2514\/6.2006-6246"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2018.2859916"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2020.2984898"},{"key":"e_1_3_2_2_7_1","unstructured":"Yilun Du Shuang Li Antonio Torralba Joshua B. Tenenbaum and Igor Mordatch. 2023. Improving Factuality and Reasoning in Language Models through Multiagent Debate. CoRR(2023)."},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"crossref","unstructured":"Moataz Medhat ElQadi Myroslava Lesiv Adrian G. Dyer and Alan Dorin. 2020. Computer vision-enhanced selection of geo-tagged photos on social network sites for land cover classification. Environ. Model. Softw.(2020) 104696.","DOI":"10.1016\/j.envsoft.2020.104696"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"crossref","unstructured":"Xueyang Feng Zhi-Yuan Chen Yujia Qin Yankai Lin Xu Chen Zhiyuan Liu and Ji-Rong Wen. 2024. Large Language Model-based Human-Agent Collaboration for Complex Task Solving. arXiv preprint arXiv:2402.12914(2024).","DOI":"10.18653\/v1\/2024.findings-emnlp.72"},{"key":"e_1_3_2_2_10_1","first-page":"10764","article-title":"PAL","author":"Gao Luyu","year":"2023","unstructured":"Luyu Gao, Aman Madaan, Shuyan Zhou, Uri Alon, Pengfei Liu, Yiming Yang, Jamie Callan, and Graham Neubig. 2023. PAL: Program-aided Language Models. In Proc. of ICML. 10764-10799.","journal-title":"Program-aided Language Models. In Proc. of ICML."},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1177\/0278364919839761"},{"key":"e_1_3_2_2_12_1","first-page":"369","article-title":"Self-supervising Fine-Grained Region Similarities for Large-Scale Image Localization","author":"Ge Yixiao","year":"2020","unstructured":"Yixiao Ge, Haibo Wang, Feng Zhu, Rui Zhao, and Hongsheng Li. 2020. Self-supervising Fine-Grained Region Similarities for Large-Scale Image Localization. In Proc. of ECCV. 369-386.","journal-title":"Proc. of ECCV."},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i1.32002"},{"key":"e_1_3_2_2_14_1","first-page":"14141","article-title":"Patch-NetVLAD","author":"Hausler Stephen","year":"2021","unstructured":"Stephen Hausler, Sourav Garg, Ming Xu, Michael Milford, and Tobias Fischer. 2021. Patch-NetVLAD: Multi-Scale Fusion of Locally-Global Descriptors for Place Recognition. In Proc. of CVPR. 14141-14152.","journal-title":"Multi-Scale Fusion of Locally-Global Descriptors for Place Recognition. In Proc. of CVPR."},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2019.2898427"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3341161.3342870"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.5244\/C.35.132"},{"key":"e_1_3_2_2_18_1","unstructured":"Gautier Izacard Patrick S. H. Lewis Maria Lomeli Lucas Hosseini Fabio Petroni Timo Schick Jane Dwivedi-Yu Armand Joulin Sebastian Riedel and Edouard Grave. 2023. Atlas: Few-shot Learning with Retrieval Augmented Language Models. J. Mach. Learn. Res.(2023) 251:1-251:43."},{"key":"e_1_3_2_2_19_1","first-page":"3","article-title":"Exploiting the Earth's Spherical Geometry to Geolocate Images","author":"Izbicki Mike","year":"2019","unstructured":"Mike Izbicki, Evangelos E. Papalexakis, and Vassilis J. Tsotras. 2019. Exploiting the Earth's Spherical Geometry to Geolocate Images. In Proc. of KDD. 3-19.","journal-title":"Proc. of KDD."},{"key":"e_1_3_2_2_20_1","first-page":"53198","article-title":"G3: an effective and adaptive framework for worldwide geolocalization using large multi-modality models","volume":"37","author":"Jia Pengyue","year":"2024","unstructured":"Pengyue Jia, Yiding Liu, Xiaopeng Li, Xiangyu Zhao, Yuhao Wang, Yantong Du, Xiao Han, Xuetao Wei, Shuaiqiang Wang, and Dawei Yin. 2024. G3: an effective and adaptive framework for worldwide geolocalization using large multi-modality models. Advances in Neural Information Processing Systems, Vol. 37 (2024), 53198-53221.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_21_1","first-page":"14165","article-title":"LLM-Blender","author":"Jiang Dongfu","year":"2023","unstructured":"Dongfu Jiang, Xiang Ren, and Bill Yuchen Lin. 2023. LLM-Blender: Ensembling Large Language Models with Pairwise Ranking and Generative Fusion. In Proc. of ACL. 14165-14178.","journal-title":"In Proc. of ACL."},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.346"},{"key":"e_1_3_2_2_23_1","volume-title":"McDonald-Maier","author":"Khaliq Ahmad","year":"2020","unstructured":"Ahmad Khaliq, Shoaib Ehsan, Zetao Chen, Michael Milford, and Klaus D. McDonald-Maier. 2020. A Holistic Visual Place Recognition Approach Using Lightweight CNNs for Significant ViewPoint and Appearance Changes. IEEE Trans. Robotics(2020), 561-569."},{"key":"e_1_3_2_2_24_1","first-page":"155","volume-title":"Leveraging EfficientNet and Contrastive Learning for Accurate Global-scale Location Estimation. In ICMR '21: International Conference on Multimedia Retrieval","author":"Kordopatis-Zilos Giorgos","year":"2021","unstructured":"Giorgos Kordopatis-Zilos, Panagiotis Galopoulos, Symeon Papadopoulos, and Ioannis Kompatsiaris. 2021. Leveraging EfficientNet and Contrastive Learning for Accurate Global-scale Location Estimation. In ICMR '21: International Conference on Multimedia Retrieval, Taipei, Taiwan, August 21-24, 2021. 155-163."},{"key":"e_1_3_2_2_25_1","first-page":"2570","article-title":"Stochastic Attraction-Repulsion Embedding for Large Scale Image Localization","author":"Liu Liu","year":"2019","unstructured":"Liu Liu, Hongdong Li, and Yuchao Dai. 2019. Stochastic Attraction-Repulsion Embedding for Large Scale Image Localization. In Proc. of ICCV. 2570-2579.","journal-title":"Proc. of ICCV."},{"key":"e_1_3_2_2_26_1","unstructured":"Zijun Liu Yanzhe Zhang Peng Li Yang Liu and Diyi Yang. 2023. Dynamic LLM-Agent Network: An LLM-agent Collaboration Framework with Agent Team Optimization. CoRR(2023)."},{"key":"e_1_3_2_2_27_1","volume-title":"Proc. of NeurIPS.","author":"Lu Pan","year":"2023","unstructured":"Pan Lu, Baolin Peng, Hao Cheng, Michel Galley, Kai-Wei Chang, Ying Nian Wu, Song-Chun Zhu, and Jianfeng Gao. 2023. Chameleon: Plug-and-Play Compositional Reasoning with Large Language Models. In Proc. of NeurIPS."},{"key":"e_1_3_2_2_28_1","volume-title":"Gallagher","author":"Luo Jiebo","year":"2011","unstructured":"Jiebo Luo, Dhiraj Joshi, Jie Yu, and Andrew C. Gallagher. 2011. Geotagging in multimedia and computer vision - a survey. Multim. Tools Appl.(2011), 187-211."},{"key":"e_1_3_2_2_29_1","first-page":"575","article-title":"Geolocation Estimation of Photos Using a Hierarchical Model and Scene Classification","author":"M\u00fcller-Budack Eric","year":"2018","unstructured":"Eric M\u00fcller-Budack, Kader Pustu-Iren, and Ralph Ewerth. 2018. Geolocation Estimation of Photos Using a Hierarchical Model and Scene Classification. In Proc. of ECCV. 575-592.","journal-title":"Proc. of ECCV."},{"key":"e_1_3_2_2_30_1","volume-title":"Proc. of NeurIPS.","author":"Ouyang Long","year":"2022","unstructured":"Long Ouyang, Jeffrey Wu, Xu Jiang, Diogo Almeida, Carroll L. Wainwright, Pamela Mishkin, Chong Zhang, Sandhini Agarwal, Katarina Slama, Alex Ray, John Schulman, Jacob Hilton, Fraser Kelton, Luke Miller, Maddie Simens, Amanda Askell, Peter Welinder, Paul F. Christiano, Jan Leike, and Ryan Lowe. 2022. Training language models to follow instructions with human feedback. In Proc. of NeurIPS."},{"key":"e_1_3_2_2_31_1","volume-title":"Francesco Montagna, Carlo Masone, and Barbara Caputo.","author":"Paolicelli Valerio","year":"2022","unstructured":"Valerio Paolicelli, Gabriele Moreno Berton, Francesco Montagna, Carlo Masone, and Barbara Caputo. 2022. Adaptive-Attentive Geolocalization From Few Queries: A Hybrid Approach. Frontiers Comput. Sci.(2022), 841817."},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48506.2021.9561812"},{"key":"e_1_3_2_2_33_1","first-page":"196","article-title":"Where in the world is this image? transformer-based geo-localization in the wild","author":"Pramanick Shraman","year":"2022","unstructured":"Shraman Pramanick, Ewa M Nowara, Joshua Gleason, Carlos D Castillo, and Rama Chellappa. 2022. Where in the world is this image? transformer-based geo-localization in the wild. In Proc. of ECCV. 196-215.","journal-title":"Proc. of ECCV."},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2018.2846566"},{"key":"e_1_3_2_2_35_1","volume-title":"Proc. of NeurIPS.","author":"Schaeffer Rylan","year":"2023","unstructured":"Rylan Schaeffer, Brando Miranda, and Sanmi Koyejo. 2023. Are Emergent Abilities of Large Language Models a Mirage?. In Proc. of NeurIPS."},{"key":"e_1_3_2_2_36_1","volume-title":"Proc. of NeurIPS.","author":"Schick Timo","year":"2023","unstructured":"Timo Schick, Jane Dwivedi-Yu, Roberto Dess\u00ec, Roberta Raileanu, Maria Lomeli, Eric Hambro, Luke Zettlemoyer, Nicola Cancedda, and Thomas Scialom. 2023. Toolformer: Language Models Can Teach Themselves to Use Tools. In Proc. of NeurIPS."},{"key":"e_1_3_2_2_37_1","first-page":"544","article-title":"CPlaNet: Enhancing Image Geolocalization by Combinatorial Partitioning of Maps","author":"Seo Paul Hongsuck","year":"2018","unstructured":"Paul Hongsuck Seo, Tobias Weyand, Jack Sim, and Bohyung Han. 2018. CPlaNet: Enhancing Image Geolocalization by Combinatorial Partitioning of Maps. In Proc. of ECCV. 544-560.","journal-title":"Proc. of ECCV."},{"key":"e_1_3_2_2_38_1","volume-title":"REPLUG: Retrieval-Augmented Black-Box Language Models. CoRR(2023).","author":"Shi Weijia","year":"2023","unstructured":"Weijia Shi, Sewon Min, Michihiro Yasunaga, Minjoon Seo, Rich James, Mike Lewis, Luke Zettlemoyer, and Wen-tau Yih. 2023. REPLUG: Retrieval-Augmented Black-Box Language Models. CoRR(2023)."},{"key":"e_1_3_2_2_39_1","volume-title":"Proc. of NeurIPS.","author":"Shinn Noah","year":"2023","unstructured":"Noah Shinn, Federico Cassano, Ashwin Gopinath, Karthik Narasimhan, and Shunyu Yao. 2023. Reflexion: language agents with verbal reinforcement learning. In Proc. of NeurIPS."},{"key":"e_1_3_2_2_40_1","volume-title":"Proc. of NeurIPS(2024)","author":"Cepeda Vicente Vivanco","year":"2024","unstructured":"Vicente Vivanco Cepeda, Gaurav Kumar Nayak, and Mubarak Shah. 2024. Geoclip: Clip-inspired alignment between locations and images for effective worldwide geo-localization. Proc. of NeurIPS(2024)."},{"key":"e_1_3_2_2_41_1","volume-title":"Cogvlm: Visual expert for pretrained language models. arXiv preprint arXiv:2311.03079(2023).","author":"Wang Weihan","year":"2023","unstructured":"Weihan Wang, Qingsong Lv, Wenmeng Yu, Wenyi Hong, Ji Qi, Yan Wang, Junhui Ji, Zhuoyi Yang, Lei Zhao, Xixuan Song, et al., 2023. Cogvlm: Visual expert for pretrained language models. arXiv preprint arXiv:2311.03079(2023)."},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"crossref","unstructured":"Xingyao Wang Sha Li and Heng Ji. 2022. Code4Struct: Code Generation for Few-Shot Structured Prediction from Natural Language. CoRR(2022).","DOI":"10.18653\/v1\/2023.acl-long.202"},{"key":"e_1_3_2_2_43_1","first-page":"2626","article-title":"Mapillary street-level sequences: A dataset for lifelong place recognition","author":"Warburg Frederik","year":"2020","unstructured":"Frederik Warburg, Soren Hauberg, Manuel Lopez-Antequera, Pau Gargallo, Yubin Kuang, and Javier Civera. 2020. Mapillary street-level sequences: A dataset for lifelong place recognition. In Proc. of CVPR. 2626-2635.","journal-title":"Proc. of CVPR."},{"key":"e_1_3_2_2_44_1","volume-title":"Proc. of NeurIPS.","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Brian Ichter, Fei Xia, Ed H. Chi, Quoc V. Le, and Denny Zhou. 2022. Chain-of-Thought Prompting Elicits Reasoning in Large Language Models. In Proc. of NeurIPS."},{"key":"e_1_3_2_2_45_1","unstructured":"Jialiang Xu Michael Moor and Jure Leskovec. 2024. Reverse Image Retrieval Cues Parametric Memory in Multimodal LLMs. arXiv preprint arXiv:2405.18740(2024)."},{"key":"e_1_3_2_2_46_1","volume-title":"Exploiting cross-modal prediction and relation consistency for semisupervised image captioning","author":"Yang Yang","year":"2022","unstructured":"Yang Yang, Hongchen Wei, Hengshu Zhu, Dianhai Yu, Hui Xiong, and Jian Yang. 2022. Exploiting cross-modal prediction and relation consistency for semisupervised image captioning. IEEE Trans. on Cybernetics(2022), 890-902."},{"key":"e_1_3_2_2_47_1","volume-title":"Proc. of ICLR.","author":"Yao Shunyu","year":"2023","unstructured":"Shunyu Yao, Jeffrey Zhao, Dian Yu, Nan Du, Izhak Shafran, Karthik R. Narasimhan, and Yuan Cao. 2023. ReAct: Synergizing Reasoning and Acting in Language Models. In Proc. of ICLR."},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"crossref","unstructured":"Mubariz Zaffar Sourav Garg Michael Milford Julian F. P. Kooij David Flynn Klaus D. McDonald-Maier and Shoaib Ehsan. 2021. VPR-Bench: An Open-Source Visual Place Recognition Evaluation Framework with Quantifiable Viewpoint and Appearance Change. Int. J. Comput. Vis.(2021) 2136-2174.","DOI":"10.1007\/s11263-021-01469-5"},{"key":"e_1_3_2_2_49_1","volume-title":"Xudong Chen, and Ben M. Chen.","author":"Zhang Lele","year":"2018","unstructured":"Lele Zhang, Fang Deng, Jie Chen, Yingcai Bi, Swee King Phang, Xudong Chen, and Ben M. Chen. 2018. Vision-Based Target Three-Dimensional Geolocation Using Unmanned Aerial Vehicles. IEEE Trans. Ind. Electron.(2018), 8052-8061."},{"key":"e_1_3_2_2_50_1","unstructured":"Chuanyang Zheng Zhengying Liu and Enze Xie et. al. 2023. Progressive-Hint Prompting Improves Reasoning in Large Language Models. CoRR(2023)."}],"event":{"name":"KDD '25: The 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Toronto ON Canada","acronym":"KDD '25","sponsor":["SIGKDD ACM Special Interest Group on Knowledge Discovery in Data","SIGMOD ACM Special Interest Group on Management of Data"]},"container-title":["Proceedings of the 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.2"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3711896.3737141","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T17:57:25Z","timestamp":1777571845000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3711896.3737141"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8,3]]},"references-count":50,"alternative-id":["10.1145\/3711896.3737141","10.1145\/3711896"],"URL":"https:\/\/doi.org\/10.1145\/3711896.3737141","relation":{},"subject":[],"published":{"date-parts":[[2025,8,3]]},"assertion":[{"value":"2025-08-03","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}