{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,28]],"date-time":"2026-07-28T23:10:41Z","timestamp":1785280241705,"version":"3.55.0"},"reference-count":69,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T00:00:00Z","timestamp":1772755200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T00:00:00Z","timestamp":1772755200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,3,6]]},"DOI":"10.1109\/wacv61042.2026.00508","type":"proceedings-article","created":{"date-parts":[[2026,5,5]],"date-time":"2026-05-05T19:59:32Z","timestamp":1778011172000},"page":"5236-5246","source":"Crossref","is-referenced-by-count":1,"title":["GAEA: A Geolocation Aware Conversational Assistant"],"prefix":"10.1109","author":[{"given":"Ron","family":"Campos","sequence":"first","affiliation":[{"name":"University of Central Florida"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ashmal","family":"Vayani","sequence":"additional","affiliation":[{"name":"University of Central Florida"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Parth Parag","family":"Kulkarni","sequence":"additional","affiliation":[{"name":"University of Central Florida"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rohit","family":"Gupta","sequence":"additional","affiliation":[{"name":"University of Central Florida"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Aizan","family":"Zafar","sequence":"additional","affiliation":[{"name":"University of Central Florida"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Aritra","family":"Dutta","sequence":"additional","affiliation":[{"name":"University of Central Florida"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mubarak","family":"Shah","sequence":"additional","affiliation":[{"name":"University of Central Florida"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","volume-title":"EarthEnv"},{"key":"ref2","volume-title":"GeoGuessr"},{"key":"ref3","volume-title":"GeoPy"},{"key":"ref4","volume-title":"Plonkit"},{"key":"ref5","volume-title":"S2-Cells"},{"key":"ref6","volume-title":"WikiMedia"},{"key":"ref7","volume-title":"WorldStandards"},{"key":"ref8","article-title":"Phi-3 technical report: A highly capable language model locally on your phone","author":"Abdin","year":"2024"},{"key":"ref9","article-title":"Gpt-4 technical report","author":"Achiam","year":"2023"},{"key":"ref10","volume-title":"Llama 3.2: Vision and edge models","year":"2024"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.52202\/068431-1723"},{"key":"ref12","article-title":"Qwen2.5-vl technical report","author":"Bai","year":"2025"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1038\/sdata.2018.214"},{"key":"ref14","first-page":"1877","article-title":"Language models are few-shot learners","author":"Brown","year":"2020","journal-title":"Advances in neural information processing systems"},{"key":"ref15","volume-title":"Geocoder: Simple, consistent","author":"Carriere"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/TIV.2022.3192102"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52733.2024.02283"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02220"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.52202\/068431-1189"},{"key":"ref20","article-title":"Gaga: Towards interactive global geolocation assistant","author":"Dou","year":"2024"},{"key":"ref21","article-title":"The Llama 3 herd of models","author":"Dubey","year":"2024"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02140"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.52202\/068431-0943"},{"key":"ref24","author":"Team","year":"2024","journal-title":"Chatglm: A family of large language models from glm-130b to glm-4 all tools"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01225"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2008.4587784"},{"key":"ref27","article-title":"Measuring mathematical problem solving with the math dataset","volume-title":"Thirty-fifth Conference on Neural Information Processing Systems Datasets and Benchmarks Track (Round 2)","author":"Hendrycks"},{"key":"ref28","article-title":"Isns: Image-specific neural style transfer for image geolocation","author":"Hong","year":"2021"},{"key":"ref29","article-title":"LoRA: Low-Rank Adaptation of Large Language Models","volume-title":"International Conference on Learning Representations","author":"Hu"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.3390\/app11146421"},{"key":"ref31","article-title":"Safe-llava: A privacy-preserving vision-language dataset and benchmark for biometric safety","author":"Kim","year":"2025"},{"key":"ref32","article-title":"Geochat: Grounded large vision-language model for remote sensing","volume-title":"The IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Kuckreja"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73036-8_17"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/MMUL.2017.9"},{"key":"ref35","author":"Li","year":"2024","journal-title":"Llava-onevision: Easy visual task transfer"},{"key":"ref36","article-title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","author":"Li","year":"2023"},{"key":"ref37","first-page":"29222","article-title":"GeoReasoner: Geo-localization with reasoning in street views using a large vision-language model","volume-title":"Proceedings of the 41st International Conference on Machine Learning","author":"Li"},{"key":"ref38","author":"Liu","year":"2024","journal-title":"LLaVA-NeXT: Improved reasoning, OCR, and world knowledge"},{"key":"ref39","first-page":"36","article-title":"Visual instruction tuning","author":"Liu","year":"2024","journal-title":"Advances in neural information processing systems"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.24894\/sprengglossarium_p00307"},{"key":"ref41","article-title":"Sb-bench: Stereotype bias benchmark for large multimodal models","author":"Narnaware","year":"2025"},{"key":"ref42","volume-title":"Gpt-4o mini: Our affordable and intelligent small model for fast, lightweight tasks","year":"2024"},{"key":"ref43","volume-title":"Openstreetmap","year":"2024"},{"key":"ref44","author":"Radford","journal-title":"Learning transferable visual models from natural language supervision"},{"issue":"8","key":"ref45","first-page":"9","article-title":"Language models are unsupervised multitask learners","volume":"1","author":"Radford","year":"2019","journal-title":"OpenAI Blog"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2888"},{"key":"ref47","doi-asserted-by":"crossref","DOI":"10.36227\/techrxiv.173834932.29831105\/v1","article-title":"Who is responsible? the data, models, users or regulations? responsible generative ai for a sustainable future","author":"Raza","year":"2025"},{"key":"ref48","article-title":"Vldbench: Vision language models disinformation detection benchmark","author":"Raza","year":"2025"},{"issue":"2","key":"ref49","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3485766","article-title":"A survey of evaluation metrics used for nlg systems","volume":"55","author":"Sai","year":"2022","journal-title":"ACM Computing Surveys (CSUR)"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01249-6_33"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/TVCG.2017.2744159"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72658-3_9"},{"key":"ref53","article-title":"Implicitqa: Going beyond frames towards implicit video reasoning","author":"Swetha","year":"2025"},{"key":"ref54","article-title":"Timelogic: A temporal logic benchmark for video qa","author":"Swetha","year":"2025"},{"key":"ref55","article-title":"Gemini: a family of highly capable multimodal models","author":"Team","year":"2023"},{"key":"ref56","article-title":"Mobillama: Towards accurate and lightweight fully transparent GPT","author":"Thawakar","year":"2024"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52734.2025.01822"},{"key":"ref58","first-page":"36","article-title":"Geoclip: Clip-inspired alignment between locations and images for effective worldwide geolocalization","author":"Cepeda","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.286"},{"key":"ref60","first-page":"2558","article-title":"Translocator: A transformer-based large-scale image geolocalization approach","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence","author":"Wang"},{"key":"ref61","author":"Wataoka","year":"2025","journal-title":"Self-preference bias in LLM-as-a-judge"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46484-8_3"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00265"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1145\/3627673.3679934"},{"key":"ref65","article-title":"Qwen2.5 technical report","author":"Yang","year":"2024"},{"key":"ref66","author":"Ye","year":"2024","journal-title":"mplug-owl3: Towards long image-sequence understanding in multi-modal large language models"},{"key":"ref67","article-title":"Justice or prejudice? quantifying biases in LLM-as-a-judge","volume-title":"The Thirteenth International Conference on Learning Representations","author":"Ye"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2017.2723009"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00123"}],"event":{"name":"2026 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)","location":"Tucson, AZ, USA","start":{"date-parts":[[2026,3,6]]},"end":{"date-parts":[[2026,3,10]]}},"container-title":["2026 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11491838\/11491925\/11492044.pdf?arnumber=11492044","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T06:11:50Z","timestamp":1778047910000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11492044\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,6]]},"references-count":69,"URL":"https:\/\/doi.org\/10.1109\/wacv61042.2026.00508","relation":{},"subject":[],"published":{"date-parts":[[2026,3,6]]}}}