{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T16:41:26Z","timestamp":1777567286062,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":30,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,29]],"date-time":"2023-10-29T00:00:00Z","timestamp":1698537600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100003399","name":"Science and Technology Commission of Shanghai Municipality","doi-asserted-by":"publisher","award":["22DZ1100803"],"award-info":[{"award-number":["22DZ1100803"]}],"id":[{"id":"10.13039\/501100003399","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,11,2]]},"DOI":"10.1145\/3607834.3616562","type":"proceedings-article","created":{"date-parts":[[2023,10,25]],"date-time":"2023-10-25T00:09:37Z","timestamp":1698192577000},"page":"31-37","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":9,"title":["Modern Backbone for Efficient Geo-localization"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8105-2336","authenticated-orcid":false,"given":"Runzhe","family":"Zhu","sequence":"first","affiliation":[{"name":"Zhejiang University &amp; Shanghai University of Engineering Science, Zhejiang, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3645-7550","authenticated-orcid":false,"given":"Mingze","family":"Yang","sequence":"additional","affiliation":[{"name":"Shanghai University of Engineering Science, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-7178-0879","authenticated-orcid":false,"given":"Kaiyu","family":"Zhang","sequence":"additional","affiliation":[{"name":"Shanghai University of Engineering Science, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-2065-5746","authenticated-orcid":false,"given":"Fei","family":"Wu","sequence":"additional","affiliation":[{"name":"Shanghai University of Engineering Science, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-5054-5108","authenticated-orcid":false,"given":"Ling","family":"Yin","sequence":"additional","affiliation":[{"name":"Shanghai University of Engineering Science, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1225-4334","authenticated-orcid":false,"given":"Yujin","family":"Zhang","sequence":"additional","affiliation":[{"name":"Shanghai University of Engineering Science, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,10,29]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Beit: Bert pre-training of image transformers. arXiv preprint arXiv:2106.08254","author":"Bao Hangbo","year":"2021","unstructured":"Hangbo Bao , Li Dong , Songhao Piao , and Furu Wei . 2021 . Beit: Bert pre-training of image transformers. arXiv preprint arXiv:2106.08254 (2021). Hangbo Bao, Li Dong, Songhao Piao, and Furu Wei. 2021. Beit: Bert pre-training of image transformers. arXiv preprint arXiv:2106.08254 (2021)."},{"key":"e_1_3_2_1_2_1","first-page":"275","article-title":"A Part-aware Attention Neural Network for Cross-view Geo-localization between UAV and Satellite","volume":"9","author":"Bui Duc Viet","year":"2022","unstructured":"Duc Viet Bui , Masao Kubo , and Hiroshi Sato . 2022 . A Part-aware Attention Neural Network for Cross-view Geo-localization between UAV and Satellite . Journal of Robotics, Networking and Artificial Life , Vol. 9 , 3 (2022), 275 -- 284 . Duc Viet Bui, Masao Kubo, and Hiroshi Sato. 2022. A Part-aware Attention Neural Network for Cross-view Geo-localization between UAV and Satellite. Journal of Robotics, Networking and Artificial Life, Vol. 9, 3 (2022), 275--284.","journal-title":"Journal of Robotics, Networking and Artificial Life"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2021.3135013"},{"key":"e_1_3_2_1_5_1","volume-title":"Sample4Geo: Hard Negative Sampling For Cross-View Geo-Localisation. arXiv preprint arXiv:2303.11851","author":"Deuser Fabian","year":"2023","unstructured":"Fabian Deuser , Konrad Habel , and Norbert Oswald . 2023. Sample4Geo: Hard Negative Sampling For Cross-View Geo-Localisation. arXiv preprint arXiv:2303.11851 ( 2023 ). Fabian Deuser, Konrad Habel, and Norbert Oswald. 2023. Sample4Geo: Hard Negative Sampling For Cross-View Geo-Localisation. arXiv preprint arXiv:2303.11851 (2023)."},{"key":"e_1_3_2_1_6_1","unstructured":"Alexey Dosovitskiy Lucas Beyer Alexander Kolesnikov Dirk Weissenborn Xiaohua Zhai Thomas Unterthiner Mostafa Dehghani Matthias Minderer Georg Heigold Sylvain Gelly etal 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020).  Alexey Dosovitskiy Lucas Beyer Alexander Kolesnikov Dirk Weissenborn Xiaohua Zhai Thomas Unterthiner Mostafa Dehghani Matthias Minderer Georg Heigold Sylvain Gelly et al. 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)."},{"key":"e_1_3_2_1_7_1","volume-title":"2023 a. Eva-02: A visual representation for neon genesis. arXiv preprint arXiv:2303.11331","author":"Fang Yuxin","year":"2023","unstructured":"Yuxin Fang , Quan Sun , Xinggang Wang , Tiejun Huang , Xinlong Wang , and Yue Cao . 2023 a. Eva-02: A visual representation for neon genesis. arXiv preprint arXiv:2303.11331 ( 2023 ). Yuxin Fang, Quan Sun, Xinggang Wang, Tiejun Huang, Xinlong Wang, and Yue Cao. 2023 a. Eva-02: A visual representation for neon genesis. arXiv preprint arXiv:2303.11331 (2023)."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01855"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-021-01453-z"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01069"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01170"},{"key":"e_1_3_2_1_13_1","volume-title":"Separable self-attention for mobile vision transformers. arXiv preprint arXiv:2206.02680","author":"Mehta Sachin","year":"2022","unstructured":"Sachin Mehta and Mohammad Rastegari . 2022. Separable self-attention for mobile vision transformers. arXiv preprint arXiv:2206.02680 ( 2022 ). Sachin Mehta and Mohammad Rastegari. 2022. Separable self-attention for mobile vision transformers. arXiv preprint arXiv:2206.02680 (2022)."},{"key":"e_1_3_2_1_14_1","volume-title":"International conference on machine learning. PMLR, 8748--8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford , Jong Wook Kim , Chris Hallacy , Aditya Ramesh , Gabriel Goh , Sandhini Agarwal , Girish Sastry , Amanda Askell , Pamela Mishkin , Jack Clark , 2021 . Learning transferable visual models from natural language supervision . In International conference on machine learning. PMLR, 8748--8763 . Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al. 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PMLR, 8748--8763."},{"key":"e_1_3_2_1_15_1","volume-title":"Contrastive representation distillation. arXiv preprint arXiv:1910.10699","author":"Tian Yonglong","year":"2019","unstructured":"Yonglong Tian , Dilip Krishnan , and Phillip Isola . 2019. Contrastive representation distillation. arXiv preprint arXiv:1910.10699 ( 2019 ). Yonglong Tian, Dilip Krishnan, and Phillip Isola. 2019. Contrastive representation distillation. arXiv preprint arXiv:1910.10699 (2019)."},{"key":"e_1_3_2_1_16_1","volume-title":"Attention is all you need. Advances in neural information processing systems","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani , Noam Shazeer , Niki Parmar , Jakob Uszkoreit , Llion Jones , Aidan N Gomez , \u0141ukasz Kaiser , and Illia Polosukhin . 2017. Attention is all you need. Advances in neural information processing systems , Vol. 30 ( 2017 ). Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems , Vol. 30 (2017)."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2021.3061265"},{"key":"e_1_3_2_1_18_1","volume-title":"Learning cross-view geo-localization embeddings via dynamic weighted decorrelation regularization. arXiv preprint arXiv:2211.05296","author":"Wang Tingyu","year":"2022","unstructured":"Tingyu Wang , Zhedong Zheng , Zunjie Zhu , Yuhan Gao , Yi Yang , and Chenggang Yan . 2022b. Learning cross-view geo-localization embeddings via dynamic weighted decorrelation regularization. arXiv preprint arXiv:2211.05296 ( 2022 ). Tingyu Wang, Zhedong Zheng, Zunjie Zhu, Yuhan Gao, Yi Yang, and Chenggang Yan. 2022b. Learning cross-view geo-localization embeddings via dynamic weighted decorrelation regularization. arXiv preprint arXiv:2211.05296 (2022)."},{"key":"e_1_3_2_1_19_1","volume-title":"Saksham Singhal, Subhojit Som, et al.","author":"Wang Wenhui","year":"2022","unstructured":"Wenhui Wang , Hangbo Bao , Li Dong , Johan Bjorck , Zhiliang Peng , Qiang Liu , Kriti Aggarwal , Owais Khan Mohammed , Saksham Singhal, Subhojit Som, et al. 2022 a. Image as a foreign language: Beit pretraining for all vision and vision-language tasks. arXiv preprint arXiv:2208.10442 (2022). Wenhui Wang, Hangbo Bao, Li Dong, Johan Bjorck, Zhiliang Peng, Qiang Liu, Kriti Aggarwal, Owais Khan Mohammed, Saksham Singhal, Subhojit Som, et al. 2022a. Image as a foreign language: Beit pretraining for all vision and vision-language tasks. arXiv preprint arXiv:2208.10442 (2022)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01385"},{"key":"e_1_3_2_1_21_1","volume-title":"Contrastive learning rivals masked image modeling in fine-tuning via feature distillation. arXiv preprint arXiv:2205.14141","author":"Wei Yixuan","year":"2022","unstructured":"Yixuan Wei , Han Hu , Zhenda Xie , Zheng Zhang , Yue Cao , Jianmin Bao , Dong Chen , and Baining Guo . 2022. Contrastive learning rivals masked image modeling in fine-tuning via feature distillation. arXiv preprint arXiv:2205.14141 ( 2022 ). Yixuan Wei, Han Hu, Zhenda Xie, Zheng Zhang, Yue Cao, Jianmin Bao, Dong Chen, and Baining Guo. 2022. Contrastive learning rivals masked image modeling in fine-tuning via feature distillation. arXiv preprint arXiv:2205.14141 (2022)."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00943"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01179"},{"key":"e_1_3_2_1_24_1","volume-title":"Dino: Detr with improved denoising anchor boxes for end-to-end object detection. arXiv preprint arXiv:2203.03605","author":"Zhang Hao","year":"2022","unstructured":"Hao Zhang , Feng Li , Shilong Liu , Lei Zhang , Hang Su , Jun Zhu , Lionel M Ni , and Heung-Yeung Shum . 2022 . Dino: Detr with improved denoising anchor boxes for end-to-end object detection. arXiv preprint arXiv:2203.03605 (2022). Hao Zhang, Feng Li, Shilong Liu, Lei Zhang, Hang Su, Jun Zhu, Lionel M Ni, and Heung-Yeung Shum. 2022. Dino: Detr with improved denoising anchor boxes for end-to-end object detection. arXiv preprint arXiv:2203.03605 (2022)."},{"key":"e_1_3_2_1_25_1","volume-title":"UAVM '23: 2023 Workshop on UAVs in Multimedia: Capturing the World from a New Perspective. In Proceedings of the 31th ACM International Conference on Multimedia Workshop.","author":"Zheng Zhedong","year":"2023","unstructured":"Zhedong Zheng , Yujiao Shi , Tingyu Wang , Jun Liu , Jianwu Fang , Yunchao Wei , and Tat-seng Chua. 2023 . UAVM '23: 2023 Workshop on UAVs in Multimedia: Capturing the World from a New Perspective. In Proceedings of the 31th ACM International Conference on Multimedia Workshop. Zhedong Zheng, Yujiao Shi, Tingyu Wang, Jun Liu, Jianwu Fang, Yunchao Wei, and Tat-seng Chua. 2023. UAVM '23: 2023 Workshop on UAVs in Multimedia: Capturing the World from a New Perspective. In Proceedings of the 31th ACM International Conference on Multimedia Workshop."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413896"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.3390\/s23020720"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2023.3249204"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00123"},{"key":"e_1_3_2_1_30_1","volume-title":"2023 a. Simple, Effective and General: A New Backbone for Cross-view Image Geo-localization. arXiv preprint arXiv:2302.01572","author":"Zhu Yingying","year":"2023","unstructured":"Yingying Zhu , Hongji Yang , Yuxin Lu , and Qiang Huang . 2023 a. Simple, Effective and General: A New Backbone for Cross-view Image Geo-localization. arXiv preprint arXiv:2302.01572 ( 2023 ). io Yingying Zhu, Hongji Yang, Yuxin Lu, and Qiang Huang. 2023 a. Simple, Effective and General: A New Backbone for Cross-view Image Geo-localization. arXiv preprint arXiv:2302.01572 (2023). io"}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","location":"Ottawa ON Canada","acronym":"MM '23","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 2023 Workshop on UAVs in Multimedia: Capturing the World from a New Perspective"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3607834.3616562","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3607834.3616562","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T16:37:05Z","timestamp":1750178225000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3607834.3616562"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,29]]},"references-count":30,"alternative-id":["10.1145\/3607834.3616562","10.1145\/3607834"],"URL":"https:\/\/doi.org\/10.1145\/3607834.3616562","relation":{},"subject":[],"published":{"date-parts":[[2023,10,29]]},"assertion":[{"value":"2023-10-29","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}