{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,19]],"date-time":"2026-06-19T08:39:55Z","timestamp":1781858395158,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":69,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["No.W2412136, No.62202445"],"award-info":[{"award-number":["No.W2412136, No.62202445"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Postgraduate Research & Practice Innovation Program of Jiangsu Province","award":["No.KYCX25_0343"],"award-info":[{"award-number":["No.KYCX25_0343"]}]},{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["No.2022YFB2901800"],"award-info":[{"award-number":["No.2022YFB2901800"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"name":"National Natural Science Foundation of China-Research Grants Council (RGC) Joint Research Scheme","award":["No.62321166652"],"award-info":[{"award-number":["No.62321166652"]}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["No.491914380007"],"award-info":[{"award-number":["No.491914380007"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]},{"name":"'GeoX' Interdisciplinary Project of Frontiers Science Center for Critical Earth Material Cycling","award":["No.20250106"],"award-info":[{"award-number":["No.20250106"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3754559","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T06:47:18Z","timestamp":1761374838000},"page":"5441-5450","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["GeoMag: A Vision-Language Model for Pixel-level Fine-Grained Remote Sensing Image Parsing"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-4909-0894","authenticated-orcid":false,"given":"Xianzhi","family":"Ma","sequence":"first","affiliation":[{"name":"Institute of Space Earth Science, School of Frontier Sciences, Nanjing University, Suzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-6253-9808","authenticated-orcid":false,"given":"Jianhui","family":"Li","sequence":"additional","affiliation":[{"name":"Institute of Space Earth Science, School of Frontier Sciences, Nanjing University, Suzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9288-4787","authenticated-orcid":false,"given":"Changhua","family":"Pei","sequence":"additional","affiliation":[{"name":"Computer Network Information Center, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0411-5396","authenticated-orcid":false,"given":"Hao","family":"Liu","sequence":"additional","affiliation":[{"name":"Institute of Space Earth Science, School of Frontier Sciences, Nanjing University, Suzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2022.3192460"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681494"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3680965"},{"key":"e_1_3_2_1_4_1","unstructured":"Jun Chen Deyao Zhu Xiaoqian Shen Xiang Li Zechun Liu Pengchuan Zhang Raghuraman Krishnamoorthi Vikas Chandra Yunyang Xiong and Mohamed Elhoseiny. 2023b. MiniGPT-v2: large language model as a unified interface for vision-language multi-task learning. (2023). arXiv preprint arXiv:2310.09478."},{"key":"e_1_3_2_1_5_1","volume-title":"Shikra: Unleashing Multimodal LLM's Referential Dialogue Magic.","author":"Chen Keqin","year":"2023","unstructured":"Keqin Chen, Zhao Zhang, Weili Zeng, Richong Zhang, Feng Zhu, and Rui Zhao. 2023a. Shikra: Unleashing Multimodal LLM's Referential Dialogue Magic. (2023). arXiv preprint arXiv:2306.15195."},{"key":"e_1_3_2_1_6_1","volume-title":"Schwing","author":"Cheng Bowen","year":"2021","unstructured":"Bowen Cheng, Anwesa Choudhuri, Ishan Misra, Alexander Kirillov, Rohit Girdhar, and Alexander G. Schwing. 2021. Mask2Former for Video Instance Segmentation. (2021). arXiv preprint arXiv:2112.10764."},{"key":"e_1_3_2_1_7_1","first-page":"1865","article-title":"Remote Sensing Image Scene Classification","volume":"105","author":"Cheng Gong","year":"2017","unstructured":"Gong Cheng, Junwei Han, and Xiaoqiang Lu. 2017. Remote Sensing Image Scene Classification: Benchmark and State of the Art. Proc. IEEE, Vol. 105, 10 (2017), 1865-1883.","journal-title":"Benchmark and State of the Art. Proc. IEEE"},{"key":"e_1_3_2_1_8_1","volume-title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding.","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2018. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. (2018). arXiv preprint arXiv:1810.04805."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"crossref","unstructured":"Xin Guo Jiangwei Lao Bo Dang Yingying Zhang Lei Yu Lixiang Ru Liheng Zhong Ziyuan Huang Kang Wu Dingxiang Hu Huimei He Jian Wang Jingdong Chen Ming Yang Yongjun Zhang and Yansheng Li. 2024. SkySense: A Multi-Modal Remote Sensing Foundation Model Towards Universal Interpretation for Earth Observation Imagery. In 2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 27662-27673.","DOI":"10.1109\/CVPR52733.2024.02613"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2024.3362475"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2023.3281334"},{"key":"e_1_3_2_1_12_1","volume-title":"LoRA: Low-Rank Adaptation of Large Language Models. In The Tenth International Conference on Learning Representations, ICLR.","author":"Hu Edward J.","year":"2022","unstructured":"Edward J. Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, and Weizhu Chen. 2022. LoRA: Low-Rank Adaptation of Large Language Models. In The Tenth International Conference on Learning Representations, ICLR."},{"key":"e_1_3_2_1_13_1","unstructured":"Jiayun Hu Yueyi He Tianyi Liang Changbo Wang and Chenhui Li. 2025. Music2Palette: Emotion-aligned Color Palette Generation via Cross-Modal Representation Learning. arXiv preprint arXiv:2507.04758."},{"key":"e_1_3_2_1_14_1","volume-title":"RSGPT: A Remote Sensing Vision Language Model and Benchmark.","author":"Hu Yuan","year":"2023","unstructured":"Yuan Hu, Jianlong Yuan, Congcong Wen, Xiaonan Lu, and Xiang Li. 2023. RSGPT: A Remote Sensing Vision Language Model and Benchmark. (2023). arXiv preprint arXiv:2307.15266."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00448"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681563"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01050"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58607-2_4"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681270"},{"key":"e_1_3_2_1_20_1","volume-title":"Segment Anything. In 2023 IEEE\/CVF International Conference on Computer Vision (ICCV). 3992-4003","author":"Kirillov Alexander","year":"2023","unstructured":"Alexander Kirillov, Eric Mintun, Nikhila Ravi, Hanzi Mao, Chloe Rolland, Laura Gustafson, Tete Xiao, Spencer Whitehead, Alexander C. Berg, Wan-Yen Lo, Piotr Doll\u00e1r, and Ross Girshick. 2023. Segment Anything. In 2023 IEEE\/CVF International Conference on Computer Vision (ICCV). 3992-4003."},{"key":"e_1_3_2_1_21_1","volume-title":"GeoChat:Grounded Large Vision-Language Model for Remote Sensing. In 2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 27831-27840","author":"Kuckreja Kartik","year":"2024","unstructured":"Kartik Kuckreja, Muhammad Sohail Danish, Muzammal Naseer, Abhijit Das, Salman Khan, and Fahad Shahbaz Khan. 2024. GeoChat:Grounded Large Vision-Language Model for Remote Sensing. In 2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 27831-27840."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00915"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.scib.2023.05.021"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00602"},{"key":"e_1_3_2_1_25_1","volume-title":"SPHINX: The Joint Mixing of Weights, Tasks, and Visual Embeddings for Multi-modal Large Language Models.","author":"Lin Ziyi","year":"2023","unstructured":"Ziyi Lin, Chris Liu, Renrui Zhang, Peng Gao, Longtian Qiu, Han Xiao, Han Qiu, Chen Lin, Wenqi Shao, Keqin Chen, Jiaming Han, Siyuan Huang, Yichi Zhang, Xuming He, Hongsheng Li, and Yu Qiao. 2023. SPHINX: The Joint Mixing of Weights, Tasks, and Visual Embeddings for Multi-modal Large Language Models. (2023). arXiv preprint arXiv:2311.07575."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2024.3478269"},{"key":"e_1_3_2_1_27_1","volume-title":"Improved Baselines with Visual Instruction Tuning. In 2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 26286-26296","author":"Liu Haotian","year":"2024","unstructured":"Haotian Liu, Chunyuan Li, Yuheng Li, and Yong Jae Lee. 2024b. Improved Baselines with Visual Instruction Tuning. In 2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 26286-26296."},{"key":"e_1_3_2_1_28_1","first-page":"4761","article-title":"Cross-Modal Progressive Comprehension for Referring Segmentation","volume":"44","author":"Liu Si","year":"2022","unstructured":"Si Liu, Tianrui Hui, Shaofei Huang, Yunchao Wei, Bo Li, and Guanbin Li. 2022. Cross-Modal Progressive Comprehension for Referring Segmentation. IEEE Transactions on Pattern Analysis and Machine Intelligence, Vol. 44, 9 (2022), 4761-4775.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"e_1_3_2_1_29_1","volume-title":"Rotated Multi-Scale Interaction Network for Referring Remote Sensing Image Segmentation. In 2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 26648-26658","author":"Liu Sihan","year":"2024","unstructured":"Sihan Liu, Yiwei Ma, Xiaoqing Zhang, Haowei Wang, Jiayi Ji, Xiaoshuai Sun, and Rongrong Ji. 2024c. Rotated Multi-Scale Interaction Network for Referring Remote Sensing Image Segmentation. In 2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 26648-26658."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2020.2988782"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2019.2951636"},{"key":"e_1_3_2_1_32_1","volume-title":"SSCNet: Synchronous Stratification and Cross-Level Feature Fusion for Road Extraction","author":"Ma Xianzhi","year":"2025","unstructured":"Xianzhi Ma, Jianhui Li, Mingyang Lv, Xiaofeng Yang, and Hao Liu. 2025. SSCNet: Synchronous Stratification and Cross-Level Feature Fusion for Road Extraction. IEEE Transactions on Intelligent Transportation Systems (2025), 1-17."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIV.2024.3393508"},{"key":"e_1_3_2_1_34_1","volume-title":"The Twelfth International Conference on Learning Representations, ICLR.","author":"Mall Utkarsh","year":"2024","unstructured":"Utkarsh Mall, Cheng Perng Phoo, Meilin Kelsey Liu, Carl Vondrick, Bharath Hariharan, and Kavita Bala. 2024. Remote Sensing Vision-Language Foundation Models without Annotations via Ground Remote Alignment. In The Twelfth International Conference on Learning Representations, ICLR."},{"key":"e_1_3_2_1_35_1","first-page":"440","article-title":"LHRS-Bot","volume":"2024","author":"Muhtar Dilxat","year":"2025","unstructured":"Dilxat Muhtar, Zhenshi Li, Feng Gu, Xueliang Zhang, and Pengfeng Xiao. 2025. LHRS-Bot: Empowering Remote Sensing with VGI-Enhanced Large Multimodal Language Model. In Computer Vision - ECCV 2024. 440-457.","journal-title":"In Computer Vision - ECCV"},{"key":"e_1_3_2_1_36_1","unstructured":"Ruizhe Ou Yuan Hu Fan Zhang Jiaxin Chen and Yu Liu. 2025. GeoPix: Multi-Modal Large Language Model for Pixel-level Image Understanding in Remote Sensing. (2025). arXiv preprint arXiv:2501.06828."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681318"},{"key":"e_1_3_2_1_38_1","volume-title":"VHM: Versatile and Honest Vision Language Model for Remote Sensing Image Analysis.","author":"Pang Chao","year":"2024","unstructured":"Chao Pang, Xingxing Weng, Jiang Wu, Jiayu Li, Yi Liu, Jiaxing Sun, Weijia Li, Shuai Wang, Litong Feng, Gui-Song Xia, and Conghui He. 2024. VHM: Versatile and Honest Vision Language Model for Remote Sensing Image Analysis. (2024). arXiv preprint arXiv:2403.20213."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CITS.2016.7546397"},{"key":"e_1_3_2_1_40_1","volume-title":"GLaMM: Pixel Grounding Large Multimodal Model. In 2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 13009-13018","author":"Rasheed Hanoona","unstructured":"Hanoona Rasheed, Muhammad Maaz, Sahal Shaji, Abdelrahman Shaker, Salman Khan, Hisham Cholakkal, Rao M. Anwer, Eric Xing, Ming-Hsuan Yang, and Fahad S. Khan. 2024. GLaMM: Pixel Grounding Large Multimodal Model. In 2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 13009-13018."},{"key":"e_1_3_2_1_41_1","volume-title":"Nicolas Carion, Chao-Yuan Wu, Ross B. Girshick, Piotr Doll\u00e1r, and Christoph Feichtenhofer.","author":"Ravi Nikhila","year":"2024","unstructured":"Nikhila Ravi, Valentin Gabeur, Yuan-Ting Hu, Ronghang Hu, Chaitanya Ryali, Tengyu Ma, Haitham Khedr, Roman R\u00e4dle, Chlo\u00e9 Rolland, Laura Gustafson, Eric Mintun, Junting Pan, Kalyan Vasudev Alwala, Nicolas Carion, Chao-Yuan Wu, Ross B. Girshick, Piotr Doll\u00e1r, and Christoph Feichtenhofer. 2024. SAM 2: Segment Anything in Images and Videos. (2024). arXiv preprint arXiv:2408.00714."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.91"},{"key":"e_1_3_2_1_43_1","volume-title":"PixelLM: Pixel Reasoning with Large Multimodal Model. In 2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 26364-26373","author":"Ren Zhongwei","year":"2024","unstructured":"Zhongwei Ren, Zhicheng Huang, Yunchao Wei, Yao Zhao, Dongmei Fu, Jiashi Feng, and Xiaojie Jin. 2024. PixelLM: Pixel Reasoning with Large Multimodal Model. In 2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 26364-26373."},{"key":"e_1_3_2_1_44_1","volume-title":"Khan","author":"Shabbir Akashah","year":"2025","unstructured":"Akashah Shabbir, Mohammed Zumri, Mohammed Bennamoun, Fahad Khan, and Salman H. Khan. 2025. GeoPixel: Pixel Grounding Large Multimodal Model in Remote Sensing. (2025). arXiv preprint arXiv:2501.13925."},{"key":"e_1_3_2_1_45_1","volume-title":"Jo a, o Magalh a, es, Devis Tuia, and Bruno Martins.","author":"Jo","year":"2024","unstructured":"Jo a, o Daniel Silva, Jo a, o Magalh a, es, Devis Tuia, and Bruno Martins. 2024. Large Language Models for Captioning and Retrieving Remote Sensing Images. (2024). arXiv preprint arXiv:2402.06475."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2020.3031111"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2023.3335484"},{"key":"e_1_3_2_1_48_1","volume-title":"SkyScript: A Large and Semantically Diverse Vision-Language Dataset for Remote Sensing. In Thirty-Eighth AAAI Conference on Artificial Intelligence, AAAI. 5805-5813","author":"Wang Zhecheng","year":"2024","unstructured":"Zhecheng Wang, Rajanie Prabha, Tianyuan Huang, Jiajun Wu, and Ram Rajagopal. 2024. SkyScript: A Large and Semantically Diverse Vision-Language Dataset for Remote Sensing. In Thirty-Eighth AAAI Conference on Artificial Intelligence, AAAI. 5805-5813."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2017.2685945"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00370"},{"key":"e_1_3_2_1_51_1","volume-title":"LAVT: Language-Aware Vision Transformer for Referring Image Segmentation. In 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 18134-18144","author":"Yang Zhao","unstructured":"Zhao Yang, Jiaqi Wang, Yansong Tang, Kai Chen, Hengshuang Zhao, and Philip H.S. Torr. 2022. LAVT: Language-Aware Vision Transformer for Referring Image Segmentation. In 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 18134-18144."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01075"},{"key":"e_1_3_2_1_53_1","unstructured":"Xinlei Yu Changmiao Wang Hui Jin Ahmed Elazab Gangyong Jia Xiang Wan Changqing Zou and Ruiquan Ge. 2025. CRISP-SAM2: SAM2 with Cross-Modal Interaction and Semantic Prompting for Multi-Organ Segmentation. (2025). arXiv preprint arXiv:2506.23121."},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3680654"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2024.3369720"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2022.3173811"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2021.3078451"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681673"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2023.3250471"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.isprsjprs.2025.01.020"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681242"},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681684"},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2025.3629885"},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2024.3510781"},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2023.3237606"},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2024.3510781"},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3680604"},{"key":"e_1_3_2_1_68_1","unstructured":"Yue Zhou Litong Feng Yiping Ke Xue Jiang Junchi Yan Xue Yang and Wayne Zhang. 2024. Towards Vision-Language Geo-Foundation Model: A Survey. (2024). arXiv preprint arXiv:2406.09385."},{"key":"e_1_3_2_1_69_1","volume-title":"The Twelfth International Conference on Learning Representations, ICLR.","author":"Zhu Deyao","year":"2024","unstructured":"Deyao Zhu, Jun Chen, Xiaoqian Shen, Xiang Li, and Mohamed Elhoseiny. 2024. MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models. In The Twelfth International Conference on Learning Representations, ICLR."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3754559","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:14:51Z","timestamp":1765340091000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3754559"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":69,"alternative-id":["10.1145\/3746027.3754559","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3754559","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}