{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T08:18:38Z","timestamp":1783153118930,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":43,"publisher":"ACM","funder":[{"name":"National Natural Science Foundation of China","award":["62302525"],"award-info":[{"award-number":["62302525"]}]},{"name":"National Natural Science Foundation of China","award":["62476107"],"award-info":[{"award-number":["62476107"]}]},{"name":"China Postdoctoral Science Foundation","award":["2025M771527"],"award-info":[{"award-number":["2025M771527"]}]},{"name":"Natural Science Foundation of Shanghai","award":["23ZR1434000"],"award-info":[{"award-number":["23ZR1434000"]}]},{"name":"Hunan Province Natural Science Foundation of China","award":["2024JJ6527"],"award-info":[{"award-number":["2024JJ6527"]}]},{"name":"National Natural Science Foundation of China","award":["62303306"],"award-info":[{"award-number":["62303306"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,13]]},"DOI":"10.1145\/3774904.3792233","type":"proceedings-article","created":{"date-parts":[[2026,4,27]],"date-time":"2026-04-27T13:28:36Z","timestamp":1777296516000},"page":"5177-5188","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Augmenting Cross-View Geo-Localization with Spatial Semantics from Vision Foundation Models"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3003-6237","authenticated-orcid":false,"given":"Ji","family":"Shen","sequence":"first","affiliation":[{"name":"School of Computer Science, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1805-0183","authenticated-orcid":false,"given":"Lixing","family":"Chen","sequence":"additional","affiliation":[{"name":"School of Computer Science, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1037-3973","authenticated-orcid":false,"given":"Yang","family":"Bai","sequence":"additional","affiliation":[{"name":"School of Automation and Intelligent Sensing, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8239-2727","authenticated-orcid":false,"given":"Zhongqi","family":"Miao","sequence":"additional","affiliation":[{"name":"School of Computer Science, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2211-2137","authenticated-orcid":false,"given":"Zhe","family":"Qu","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, Central South University, Changsha, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8629-4622","authenticated-orcid":false,"given":"Pan","family":"Zhou","sequence":"additional","affiliation":[{"name":"School of Cyber Science and Engineering, Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6831-3973","authenticated-orcid":false,"given":"Jianhua","family":"Li","sequence":"additional","affiliation":[{"name":"School of Computer Science, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,4,12]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV61041.2025.00523"},{"key":"e_1_3_2_1_2_1","volume-title":"Genie: Generative Interactive Environments. arXiv preprint arXiv:2402.15391","author":"Bruce Jake","year":"2024","unstructured":"Jake Bruce, Michael Dennis, Ashley Edwards, Jack Parker-Holder, Yuge Shi, Edward Hughes, Matthew Lai, Aditi Mavalankar, Richie Steigerwald, Chris Apps, et al., 2024. Genie: Generative Interactive Environments. arXiv preprint arXiv:2402.15391 (2024). https:\/\/arxiv.org\/abs\/2402.15391"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01545"},{"key":"e_1_3_2_1_4_1","volume-title":"World Models. arXiv preprint arXiv:1803.10122","author":"Ha David","year":"2018","unstructured":"David Ha and J\u00fcrgen Schmidhuber. 2018. World Models. arXiv preprint arXiv:1803.10122 (2018). https:\/\/arxiv.org\/abs\/1803.10122"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00758"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00758"},{"key":"e_1_3_2_1_7_1","unstructured":"Max Jaderberg Karen Simonyan Andrew Zisserman and Koray Kavukcuoglu. 2015. Spatial Transformer Networks. In Advances in Neural Information Processing Systems (NeurIPS). https:\/\/arxiv.org\/abs\/1506.02025"},{"key":"e_1_3_2_1_8_1","volume-title":"Understanding Dimensional Collapse in Contrastive Self-supervised Learning. In International Conference on Learning Representations (ICLR). https:\/\/openreview.net\/forum?id=YevsQ05DEN7","author":"Jing Li","year":"2022","unstructured":"Li Jing, Pascal Vincent, Yann LeCun, and Yuandong Tian. 2022. Understanding Dimensional Collapse in Contrastive Self-supervised Learning. In International Conference on Learning Representations (ICLR). https:\/\/openreview.net\/forum?id=YevsQ05DEN7"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.3390\/s25154678"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01582"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7299135"},{"key":"e_1_3_2_1_12_1","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition Workshops. 1-7.","author":"Lin Tsung-Yi","unstructured":"Tsung-Yi Lin, James Hays, and Alexei A. Efros. 2013. Cross-view image geolocalization. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition Workshops. 1-7."},{"key":"e_1_3_2_1_13_1","volume-title":"Visual Instruction Tuning. Advances in Neural Information Processing Systems (NeurIPS)","author":"Liu Haotian","year":"2024","unstructured":"Haotian Liu, Chunyuan Li, Qingyang Wu, and Yong Jae Lee. 2024. Visual Instruction Tuning. Advances in Neural Information Processing Systems (NeurIPS) (2024). https:\/\/arxiv.org\/abs\/2304.08485"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00577"},{"key":"e_1_3_2_1_15_1","volume-title":"PETRv2: A Unified Framework for 3D Perception from Multi-View Images. arXiv preprint arXiv:2206.01256","author":"Liu Yiming","year":"2022","unstructured":"Yiming Liu, Tianwei Wang, Xinggang Li, Xiaohu Zhang, Jianbin Li, and Huchuan Lu. 2022b. PETRv2: A Unified Framework for 3D Perception from Multi-View Images. arXiv preprint arXiv:2206.01256 (2022)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01167"},{"key":"e_1_3_2_1_17_1","volume-title":"Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101","author":"Loshchilov Ilya","year":"2017","unstructured":"Ilya Loshchilov and Frank Hutter. 2017. Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101 (2017)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58568-6_12"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00592"},{"key":"e_1_3_2_1_20_1","volume-title":"Learning Transferable Visual Models From Natural Language Supervision. In International Conference on Machine Learning (ICML). https:\/\/arxiv.org\/abs\/2103","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al., 2021. Learning Transferable Visual Models From Natural Language Supervision. In International Conference on Machine Learning (ICML). https:\/\/arxiv.org\/abs\/2103.00020"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA46639.2022.9811901"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNN.2008.2005605"},{"key":"e_1_3_2_1_23_1","volume-title":"Advances in Neural Information Processing Systems","volume":"32","author":"Shi Yujiao","year":"2019","unstructured":"Yujiao Shi, Liu Liu, Xin Yu, and Hongdong Li. 2019. Spatial-aware feature aggregation for image based cross-view geo-localization. Advances in Neural Information Processing Systems, Vol. 32 (2019)."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00412"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.216"},{"key":"e_1_3_2_1_26_1","volume-title":"Understanding the Behaviour of Contrastive Loss. In IEEE Conference on Computer Vision and Pattern Recognition (CVPR). 2495-2504","author":"Wang Feng","year":"2021","unstructured":"Feng Wang and Huaping Liu. 2021. Understanding the Behaviour of Contrastive Loss. In IEEE Conference on Computer Vision and Pattern Recognition (CVPR). 2495-2504."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2021.3061265"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00864"},{"key":"e_1_3_2_1_29_1","volume-title":"On the Theoretical Limitations of Embedding-Based Retrieval. arXiv preprint arXiv:2508.21038","author":"Weller Orion","year":"2025","unstructured":"Orion Weller, Michael Boratko, Iftekhar Naim, and Jinhyuk Lee. 2025. On the Theoretical Limitations of Embedding-Based Retrieval. arXiv preprint arXiv:2508.21038 (2025)."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2015.7301385"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.451"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.21105\/joss.05663"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.3934\/era.2023210"},{"key":"e_1_3_2_1_34_1","first-page":"29009","article-title":"Cross-view geo-localization with layer-to-layer transformer","volume":"34","author":"Yang Hongji","year":"2021","unstructured":"Hongji Yang, Xiufan Lu, and Yingying Zhu. 2021. Cross-view geo-localization with layer-to-layer transformer. Advances in Neural Information Processing Systems, Vol. 34 (2021), 29009-29020.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_35_1","volume-title":"SkyDiffusion: Street-to-Satellite Image Synthesis with Diffusion Models and BEV Paradigm. arXiv preprint arXiv:2408.01812","author":"Ye Junyan","year":"2024","unstructured":"Junyan Ye, Jun He, Weijia Li, Zhutao Lv, Jinhua Yu, Haote Yang, and Conghui He. 2024. SkyDiffusion: Street-to-Satellite Image Synthesis with Diffusion Models and BEV Paradigm. arXiv preprint arXiv:2408.01812 (2024)."},{"key":"e_1_3_2_1_36_1","volume-title":"GeoDTR: toward generic cross-view geolocalization via geometric disentanglement","author":"Zhang Xiaohan","year":"2024","unstructured":"Xiaohan Zhang, Xingyu Li, Waqas Sultani, Chen Chen, and Safwan Wshah. 2024. GeoDTR: toward generic cross-view geolocalization via geometric disentanglement. IEEE Transactions on Pattern Analysis and Machine Intelligence (2024)."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i3.25457"},{"key":"e_1_3_2_1_38_1","first-page":"1","article-title":"SSA-Net: Spatial scale attention network for image-based geo-localization","volume":"19","author":"Zhang Xiuwei","year":"2021","unstructured":"Xiuwei Zhang, Xiangchuang Meng, Hanlin Yin, Yixin Wang, Yuanzeng Yue, Yinghui Xing, and Yanning Zhang. 2021. SSA-Net: Spatial scale attention network for image-based geo-localization. IEEE Geoscience and Remote Sensing Letters, Vol. 19 (2021), 1-5.","journal-title":"IEEE Geoscience and Remote Sensing Letters"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00123"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV48630.2021.00080"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00364"},{"key":"e_1_3_2_1_42_1","first-page":"1","article-title":"Geographic semantic network for cross-view image geo-localization","volume":"60","author":"Zhu Yingying","year":"2021","unstructured":"Yingying Zhu, Bin Sun, Xiufan Lu, and Sen Jia. 2021a. Geographic semantic network for cross-view image geo-localization. IEEE Transactions on Geoscience and Remote Sensing, Vol. 60 (2021), 1-15.","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"e_1_3_2_1_43_1","volume-title":"effective and general: A new backbone for cross-view image geo-localization. arXiv preprint arXiv:2302.01572","author":"Zhu Yingying","year":"2023","unstructured":"Yingying Zhu, Hongji Yang, Yuxin Lu, and Qiang Huang. 2023. Simple, effective and general: A new backbone for cross-view image geo-localization. arXiv preprint arXiv:2302.01572 (2023)."}],"event":{"name":"WWW '26: The ACM Web Conference 2026","location":"Dubai United Arab Emirates","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"]},"container-title":["Proceedings of the ACM Web Conference 2026"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3774904.3792233","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T07:48:50Z","timestamp":1783151330000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774904.3792233"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,12]]},"references-count":43,"alternative-id":["10.1145\/3774904.3792233","10.1145\/3774904"],"URL":"https:\/\/doi.org\/10.1145\/3774904.3792233","relation":{},"subject":[],"published":{"date-parts":[[2026,4,12]]},"assertion":[{"value":"2026-04-12","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}