{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T10:35:14Z","timestamp":1783074914370,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":37,"publisher":"ACM","funder":[{"name":"Key Research and Development Program of Zhejiang Province","award":["2025C01026"],"award-info":[{"award-number":["2025C01026"]}]},{"name":"Ningbo Yongjiang Talent Introduction Programme","award":["2023A-397-G"],"award-info":[{"award-number":["2023A-397-G"]}]},{"name":"Young Elite Scientists Sponsorship Program by CAST","award":["2024QNRC001"],"award-info":[{"award-number":["2024QNRC001"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62402429, U24A20326, 62441236"],"award-info":[{"award-number":["62402429, U24A20326, 62441236"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755646","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:27:39Z","timestamp":1761377259000},"page":"11648-11656","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Evaluating the Robustness of Multimodal Agents Against Active Environmental Injection Attacks"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-7088-7414","authenticated-orcid":false,"given":"Yurun","family":"Chen","sequence":"first","affiliation":[{"name":"School of Software Technology, Zhejiang University, Hangzhou, ZHejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-6055-8996","authenticated-orcid":false,"given":"Xueyu","family":"Hu","sequence":"additional","affiliation":[{"name":"College of Computer Science and Technology, Zhejiang University, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9674-4132","authenticated-orcid":false,"given":"Keting","family":"Yin","sequence":"additional","affiliation":[{"name":"School of Software Technology, Zhejiang University, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2258-1291","authenticated-orcid":false,"given":"Juncheng","family":"Li","sequence":"additional","affiliation":[{"name":"School of Software Technology, Zhejiang University, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0030-8289","authenticated-orcid":false,"given":"Shengyu","family":"Zhang","sequence":"additional","affiliation":[{"name":"School of Software Technology, Zhejiang University, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Android Developers. 2025. Notifications on the Home Screen. https:\/\/developer.android.google.cn\/design\/ui\/mobile\/guides\/home-screen\/notifications Accessed: 2025-02-13."},{"key":"e_1_3_2_1_2_1","unstructured":"Anthropic. 2024. Introducing computer use a new Claude 3.5 Sonnet and Claude 3.5 Haiku. https:\/\/www.anthropic.com\/news\/3-5-models-and-computer-use Accessed: 2025-01-08."},{"key":"e_1_3_2_1_3_1","unstructured":"Apple Inc. 2024. Apple Intelligence is available today on iPhone iPad and Mac. https:\/\/www.apple.com\/sg\/newsroom\/2024\/10\/apple-intelligence-is-available-today-on-iphone-ipad-and-mac\/ Accessed: 2025-01-08."},{"key":"e_1_3_2_1_4_1","volume-title":"Qwen-vl: A frontier large vision-language model with versatile abilities. arXiv preprint arXiv:2308.12966","author":"Bai Jinze","year":"2023","unstructured":"Jinze Bai, Shuai Bai, Shusheng Yang, Shijie Wang, Sinan Tan, Peng Wang, Junyang Lin, Chang Zhou, and Jingren Zhou. 2023. Qwen-vl: A frontier large vision-language model with versatile abilities. arXiv preprint arXiv:2308.12966 (2023)."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"crossref","unstructured":"Kanzhi Cheng Qiushi Sun Yougang Chu Fangzhi Xu Yantao Li Jianbing Zhang and Zhiyong Wu. 2024. SeeClick: Harnessing GUI Grounding for Advanced Visual GUI Agents. arXiv:2401.10935 [cs.HC] https:\/\/arxiv.org\/abs\/2401.10935","DOI":"10.18653\/v1\/2024.acl-long.505"},{"key":"e_1_3_2_1_6_1","unstructured":"Yang Deng Xuan Zhang Wenxuan Zhang Yifei Yuan See-Kiong Ng and Tat-Seng Chua. 2024. On the Multi-turn Instruction Following for Conversational Web Agents. arXiv:2402.15057 [cs.CL] https:\/\/arxiv.org\/abs\/2402.15057"},{"key":"e_1_3_2_1_7_1","volume-title":"Bidipta Sarkar, Rohan Taori, Yusuke Noda, Demetri Terzopoulos, Yejin Choi, et al.","author":"Durante Zane","year":"2024","unstructured":"Zane Durante, Qiuyuan Huang, Naoki Wake, Ran Gong, Jae Sung Park, Bidipta Sarkar, Rohan Taori, Yusuke Noda, Demetri Terzopoulos, Yejin Choi, et al., 2024. Agent ai: Surveying the horizons of multimodal interaction. arXiv preprint arXiv:2401.03568 (2024)."},{"key":"e_1_3_2_1_8_1","unstructured":"Google. 2024. Google introduces Gemini 2.0: A new AI model for the agentic era. https:\/\/blog.google\/technology\/google-deepmind\/google-gemini-ai-update-december-2024\/#ceo-message Accessed: 2025-01-08."},{"key":"e_1_3_2_1_9_1","volume-title":"Mustafa Safdari, Yutaka Matsuo, Douglas Eck, and Aleksandra Faust.","author":"Gur Izzeddin","year":"2024","unstructured":"Izzeddin Gur, Hiroki Furuta, Austin Huang, Mustafa Safdari, Yutaka Matsuo, Douglas Eck, and Aleksandra Faust. 2024. A Real-World WebAgent with Planning, Long Context Understanding, and Program Synthesis. arXiv:2307.12856 [cs.LG] https:\/\/arxiv.org\/abs\/2307.12856"},{"key":"e_1_3_2_1_10_1","unstructured":"Wenyi Hong Weihan Wang Ming Ding Wenmeng Yu Qingsong Lv Yan Wang Yean Cheng Shiyu Huang Junhui Ji Zhao Xue et al. 2024. Cogvlm2: Visual language models for image and video understanding. arXiv preprint arXiv:2408.16500 (2024)."},{"key":"e_1_3_2_1_11_1","unstructured":"Jakub Hoscilowicz Bartosz Maj Bartosz Kozakiewicz Oleksii Tymoshchuk and Artur Janicki. 2024. ClickAgent: Enhancing UI Location Capabilities of Autonomous Agents. arXiv:2410.11872 [cs.HC] https:\/\/arxiv.org\/abs\/2410.11872"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.20944\/preprints202412.2294.v1"},{"key":"e_1_3_2_1_13_1","volume-title":"Infiagent-dabench: Evaluating agents on data analysis tasks. arXiv preprint arXiv:2401.05507","author":"Hu Xueyu","year":"2024","unstructured":"Xueyu Hu, Ziyu Zhao, Shuang Wei, Ziwei Chai, Qianli Ma, Guoyin Wang, Xuwu Wang, Jing Su, Jingjing Xu, Ming Zhu, et al., 2024b. Infiagent-dabench: Evaluating agents on data analysis tasks. arXiv preprint arXiv:2401.05507 (2024)."},{"key":"e_1_3_2_1_14_1","unstructured":"Aaron Hurst Adam Lerer Adam P Goucher Adam Perelman Aditya Ramesh Aidan Clark AJ Ostrow Akila Welihinda Alan Hayes Alec Radford et al. 2024. Gpt-4o system card. arXiv preprint arXiv:2410.21276 (2024)."},{"key":"e_1_3_2_1_15_1","volume-title":"Hojun Choi, Steven Y. Ko, Sangeun Oh, and Insik Shin.","author":"Lee Sunjae","year":"2024","unstructured":"Sunjae Lee, Junyoung Choi, Jungjae Lee, Munim Hasan Wasi, Hojun Choi, Steven Y. Ko, Sangeun Oh, and Insik Shin. 2024. Explore, Select, Derive, and Recall: Augmenting LLM with Human-like Memory for Mobile Task Automation. arXiv:2312.03003 [cs.HC] https:\/\/arxiv.org\/abs\/2312.03003"},{"key":"e_1_3_2_1_16_1","volume-title":"Llava-onevision: Easy visual task transfer. arXiv preprint arXiv:2408.03326","author":"Li Bo","year":"2024","unstructured":"Bo Li, Yuanhan Zhang, Dong Guo, Renrui Zhang, Feng Li, Hao Zhang, Kaichen Zhang, Peiyuan Zhang, Yanwei Li, Ziwei Liu, et al., 2024. Llava-onevision: Easy visual task transfer. arXiv preprint arXiv:2408.03326 (2024)."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"crossref","unstructured":"Tao Li Gang Li Zhiwei Deng Bryan Wang and Yang Li. 2023. A Zero-Shot Language Agent for Computer Control with Structured Reflection. arXiv:2310.08740 [cs.CL] https:\/\/arxiv.org\/abs\/2310.08740","DOI":"10.18653\/v1\/2023.findings-emnlp.753"},{"key":"e_1_3_2_1_18_1","volume-title":"EIA: Environmental Injection Attack on Generalist Web Agents for Privacy Leakage. arXiv:2409.11295 [cs.CR] https:\/\/arxiv.org\/abs\/2409.11295","author":"Liao Zeyi","year":"2024","unstructured":"Zeyi Liao, Lingbo Mo, Chejian Xu, Mintong Kang, Jiawei Zhang, Chaowei Xiao, Yuan Tian, Bo Li, and Huan Sun. 2024. EIA: Environmental Injection Attack on Generalist Web Agents for Privacy Leakage. arXiv:2409.11295 [cs.CR] https:\/\/arxiv.org\/abs\/2409.11295"},{"key":"e_1_3_2_1_19_1","volume-title":"InfiGUIAgent: A Multimodal Generalist GUI Agent with Native Reasoning and Reflection. arXiv preprint arXiv:2501.04575","author":"Liu Yuhang","year":"2025","unstructured":"Yuhang Liu, Pengxiang Li, Zishu Wei, Congkai Xie, Xueyu Hu, Xinchen Xu, Shengyu Zhang, Xiaotian Han, Hongxia Yang, and Fei Wu. 2025. InfiGUIAgent: A Multimodal Generalist GUI Agent with Native Reasoning and Reflection. arXiv preprint arXiv:2501.04575 (2025)."},{"key":"e_1_3_2_1_20_1","unstructured":"Xinbei Ma Yiting Wang Yao Yao Tongxin Yuan Aston Zhang Zhuosheng Zhang and Hai Zhao. 2024. Caution for the Environment: Multimodal Agents are Susceptible to Environmental Distractions. arXiv:2408.02544 [cs.CL] https:\/\/arxiv.org\/abs\/2408.02544"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"crossref","unstructured":"Dang Nguyen Jian Chen Yu Wang Gang Wu Namyong Park Zhengmian Hu Hanjia Lyu Junda Wu Ryan Aponte Yu Xia Xintong Li Jing Shi Hongjie Chen Viet Dac Lai Zhouhang Xie Sungchul Kim Ruiyi Zhang Tong Yu Mehrab Tanjim Nesreen K. Ahmed Puneet Mathur Seunghyun Yoon Lina Yao Branislav Kveton Thien Huu Nguyen Trung Bui Tianyi Zhou Ryan A. Rossi and Franck Dernoncourt. 2024. GUI Agents: A Survey. arXiv:2412.13501 [cs.AI] https:\/\/arxiv.org\/abs\/2412.13501","DOI":"10.18653\/v1\/2025.findings-acl.1158"},{"key":"e_1_3_2_1_22_1","unstructured":"OpenAI. 2025. Introducing Operator. https:\/\/openai.com\/index\/introducing-operator\/ Accessed: 2024-1-23."},{"key":"e_1_3_2_1_23_1","unstructured":"Christopher Rawles Sarah Clinckemaillie Yifan Chang Jonathan Waltz Gabrielle Lau Marybeth Fair Alice Li William Bishop Wei Li Folawiyo Campbell-Ajala et al. 2024. AndroidWorld: A dynamic benchmarking environment for autonomous agents. arXiv preprint arXiv:2405.14573 (2024)."},{"key":"e_1_3_2_1_24_1","volume-title":"Cradle: Empowering Foundation Agents Towards General Computer Control. arXiv:2403.03186 [cs.AI] https:\/\/arxiv.org\/abs\/2403.03186","author":"Tan Weihao","year":"2024","unstructured":"Weihao Tan, Wentao Zhang, Xinrun Xu, Haochong Xia, Ziluo Ding, Boyu Li, Bohan Zhou, Junpeng Yue, Jiechuan Jiang, Yewen Li, Ruyi An, Molei Qin, Chuqiao Zong, Longtao Zheng, Yujie Wu, Xiaoqiang Chai, Yifei Bi, Tianbao Xie, Pengjie Gu, Xiyun Li, Ceyao Zhang, Long Tian, Chaojie Wang, Xinrun Wang, B\u00f6rje F. Karlsson, Bo An, Shuicheng Yan, and Zongqing Lu. 2024. Cradle: Empowering Foundation Agents Towards General Computer Control. arXiv:2403.03186 [cs.AI] https:\/\/arxiv.org\/abs\/2403.03186"},{"key":"e_1_3_2_1_25_1","unstructured":"Junyang Wang Haiyang Xu Haitao Jia Xi Zhang Ming Yan Weizhou Shen Ji Zhang Fei Huang and Jitao Sang. 2024b. Mobile-Agent-v2: Mobile Device Operation Assistant with Effective Navigation via Multi-Agent Collaboration. arXiv:2406.01014 [cs.CL] https:\/\/arxiv.org\/abs\/2406.01014"},{"key":"e_1_3_2_1_26_1","unstructured":"Junyang Wang Haiyang Xu Jiabo Ye Ming Yan Weizhou Shen Ji Zhang Fei Huang and Jitao Sang. 2024c. Mobile-Agent: Autonomous Multi-Modal Mobile Device Agent with Visual Perception. arXiv:2401.16158 [cs.CL] https:\/\/arxiv.org\/abs\/2401.16158"},{"key":"e_1_3_2_1_27_1","unstructured":"Peng Wang Shuai Bai Sinan Tan Shijie Wang Zhihao Fan Jinze Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge et al. 2024a. Qwen2-vl: Enhancing vision-language model's perception of the world at any resolution. arXiv preprint arXiv:2409.12191 (2024)."},{"key":"e_1_3_2_1_28_1","volume-title":"Ruslan Salakhutdinov, Daniel Fried, and Aditi Raghunathan.","author":"Wu Chen Henry","year":"2024","unstructured":"Chen Henry Wu, Rishi Shah, Jing Yu Koh, Ruslan Salakhutdinov, Daniel Fried, and Aditi Raghunathan. 2024b. Dissecting Adversarial Robustness of Multimodal LM Agents. arXiv:2406.12814 [cs.LG] https:\/\/arxiv.org\/abs\/2406.12814"},{"key":"e_1_3_2_1_29_1","volume-title":"WIPI: A New Web Threat for LLM-Driven Web Agents. arXiv:2402.16965 [cs.CR] https:\/\/arxiv.org\/abs\/2402.16965","author":"Wu Fangzhou","year":"2024","unstructured":"Fangzhou Wu, Shutong Wu, Yulong Cao, and Chaowei Xiao. 2024c. WIPI: A New Web Threat for LLM-Driven Web Agents. arXiv:2402.16965 [cs.CR] https:\/\/arxiv.org\/abs\/2402.16965"},{"key":"e_1_3_2_1_30_1","unstructured":"Zhiyong Wu Chengcheng Han Zichen Ding Zhenmin Weng Zhoumianze Liu Shunyu Yao Tao Yu and Lingpeng Kong. 2024a. OS-Copilot: Towards Generalist Computer Agents with Self-Improvement. arXiv:2402.07456 [cs.AI] https:\/\/arxiv.org\/abs\/2402.07456"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11432-024-4222-0"},{"key":"e_1_3_2_1_32_1","unstructured":"An Yan Zhengyuan Yang Wanrong Zhu Kevin Lin Linjie Li Jianfeng Wang Jianwei Yang Yiwu Zhong Julian McAuley Jianfeng Gao Zicheng Liu and Lijuan Wang. 2023. GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation. arXiv:2311.07562 [cs.CV] https:\/\/arxiv.org\/abs\/2311.07562"},{"key":"e_1_3_2_1_33_1","unstructured":"Jianwei Yang Hao Zhang Feng Li Xueyan Zou Chunyuan Li and Jianfeng Gao. 2023. Set-of-Mark Prompting Unleashes Extraordinary Visual Grounding in GPT-4V. arXiv:2310.11441 [cs.CV] https:\/\/arxiv.org\/abs\/2310.11441"},{"key":"e_1_3_2_1_34_1","unstructured":"Yulong Yang Xinshan Yang Shuaidong Li Chenhao Lin Zhengyu Zhao Chao Shen and Tianwei Zhang. 2024. Security Matrix for Multimodal Agents on Mobile Devices: A Systematic and Proof of Concept Study. arXiv:2407.09295 [cs.CR] https:\/\/arxiv.org\/abs\/2407.09295"},{"key":"e_1_3_2_1_35_1","unstructured":"Chi Zhang Zhao Yang Jiaxuan Liu Yucheng Han Xin Chen Zebiao Huang Bin Fu and Gang Yu. 2023. AppAgent: Multimodal Agents as Smartphone Users. arXiv:2312.13771 [cs.CV] https:\/\/arxiv.org\/abs\/2312.13771"},{"key":"e_1_3_2_1_36_1","unstructured":"Yanzhe Zhang Tao Yu and Diyi Yang. 2024. Attacking Vision-Language Computer Agents via Pop-ups. arXiv:2411.02391 [cs.CL] https:\/\/arxiv.org\/abs\/2411.02391"},{"key":"e_1_3_2_1_37_1","unstructured":"Boyuan Zheng Boyu Gou Jihyung Kil Huan Sun and Yu Su. 2024. GPT-4V(ision) is a Generalist Web Agent if Grounded. arXiv:2401.01614 [cs.IR] https:\/\/arxiv.org\/abs\/2401.01614"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755646","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:56:20Z","timestamp":1765342580000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755646"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":37,"alternative-id":["10.1145\/3746027.3755646","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755646","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}