{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T08:18:02Z","timestamp":1783153082323,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":56,"publisher":"ACM","funder":[{"name":"National Natural Science Foundation of China","award":["62322603"],"award-info":[{"award-number":["62322603"]}]},{"name":"National Natural Science Foundation of China","award":["62502310"],"award-info":[{"award-number":["62502310"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,13]]},"DOI":"10.1145\/3774904.3792215","type":"proceedings-article","created":{"date-parts":[[2026,4,27]],"date-time":"2026-04-27T13:28:36Z","timestamp":1777296516000},"page":"7024-7035","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["ColorBench: Benchmarking Mobile Agents with Graph-Structured Framework for Complex Long-Horizon Tasks"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-7731-1332","authenticated-orcid":false,"given":"Yuanyi","family":"Song","sequence":"first","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-0320-3629","authenticated-orcid":false,"given":"Heyuan","family":"Huang","sequence":"additional","affiliation":[{"name":"OPPO, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8587-1651","authenticated-orcid":false,"given":"Qiqiang","family":"Lin","sequence":"additional","affiliation":[{"name":"OPPO, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-5842-129X","authenticated-orcid":false,"given":"Yin","family":"Zhao","sequence":"additional","affiliation":[{"name":"OPPO, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-4449-522X","authenticated-orcid":false,"given":"Xiangmou","family":"Qu","sequence":"additional","affiliation":[{"name":"OPPO, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-2577-6221","authenticated-orcid":false,"given":"Jun","family":"Wang","sequence":"additional","affiliation":[{"name":"OPPO, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-3180-0668","authenticated-orcid":false,"given":"Xingyu","family":"Lou","sequence":"additional","affiliation":[{"name":"OPPO, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9148-3997","authenticated-orcid":false,"given":"Weiwen","family":"Liu","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4183-3645","authenticated-orcid":false,"given":"Zhuosheng","family":"Zhang","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0481-5341","authenticated-orcid":false,"given":"Jun","family":"Wang","sequence":"additional","affiliation":[{"name":"OPPO, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-4925-7703","authenticated-orcid":false,"given":"Zhaoxiang","family":"Wang","sequence":"additional","affiliation":[{"name":"OPPO, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4457-2820","authenticated-orcid":false,"given":"Yong","family":"Yu","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0127-2425","authenticated-orcid":false,"given":"Weinan","family":"Zhang","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,4,12]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Zhipu AI. 2024. GLM-4.5V Technical Report. https:\/\/zhipu.ai\/en\/blog\/glm-4-5v. Accessed: 2024-12-19."},{"key":"e_1_3_2_1_2_1","unstructured":"Jinze Bai Shuai Bai Shusheng Yang Shijie Wang Sinan Tan Peng Wang Junyang Lin Chang Zhou and Jingren Zhou. 2023. Qwen-VL: A Versatile Vision-Language Model for Understanding Localization Text Reading and Beyond. arXiv:2308.12966 [cs.CV] https:\/\/arxiv.org\/abs\/2308.12966"},{"key":"e_1_3_2_1_3_1","unstructured":"Shuai Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge Sibo Song Kai Dang Peng Wang Shijie Wang Jun Tang et al. 2025. Qwen2. 5-vl technical report. arXiv preprint arXiv:2502.13923 (2025)."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMC.2018.2868932"},{"key":"e_1_3_2_1_5_1","volume-title":"Less is more: Empowering gui agent with context-aware simplification. arXiv preprint arXiv:2507.03730","author":"Chen Gongwei","year":"2025","unstructured":"Gongwei Chen, Xurui Zhou, Rui Shao, Yibo Lyu, Kaiwen Zhou, Shuai Wang, Wentao Li, Yinchuan Li, Zhongang Qi, and Liqiang Nie. 2025c. Less is more: Empowering gui agent with context-aware simplification. arXiv preprint arXiv:2507.03730 (2025)."},{"key":"e_1_3_2_1_6_1","volume-title":"NeurIPS 2024 Workshop on Open-World Agents.","author":"Chen Jingxuan","year":"2024","unstructured":"Jingxuan Chen, Derek Yuen, Bin Xie, Yuhao Yang, Gongwei Chen, Zhihao Wu, Li Yixing, Xurui Zhou, Weiwen Liu, Shuai Wang, et al., 2024b. Spa-bench: A comprehensive benchmark for smartphone agent evaluation. In NeurIPS 2024 Workshop on Open-World Agents."},{"key":"e_1_3_2_1_7_1","volume-title":"Guicourse: From general vision language models to versatile gui agents. arXiv preprint arXiv:2406.11317","author":"Chen Wentong","year":"2024","unstructured":"Wentong Chen, Junbo Cui, Jinyi Hu, Yujia Qin, Junjie Fang, Yue Zhao, Chongyi Wang, Jun Liu, Guirong Chen, Yupeng Huo, et al., 2024a. Guicourse: From general vision language models to versatile gui agents. arXiv preprint arXiv:2406.11317 (2024)."},{"key":"e_1_3_2_1_8_1","volume-title":"PG-Agent: An Agent Powered by Page Graph. arXiv preprint arXiv:2509.03536","author":"Chen Weizhi","year":"2025","unstructured":"Weizhi Chen, Ziwei Wang, Leyang Yang, Sheng Zhou, Xiaoxuan Tang, Jiajun Bu, Yong Li, and Wei Jiang. 2025b. PG-Agent: An Agent Powered by Page Graph. arXiv preprint arXiv:2509.03536 (2025)."},{"key":"e_1_3_2_1_9_1","volume-title":"HarmonyGuard: Toward Safety and Utility in Web Agents via Adaptive Policy Enhancement and Dual-Objective Optimization. arXiv preprint arXiv:2508.04010","author":"Chen Yurun","year":"2025","unstructured":"Yurun Chen, Xavier Hu, Yuhan Liu, Keting Yin, Juncheng Li, Zhuosheng Zhang, and Shengyu Zhang. 2025a. HarmonyGuard: Toward Safety and Utility in Web Agents via Adaptive Policy Enhancement and Dual-Objective Optimization. arXiv preprint arXiv:2508.04010 (2025)."},{"key":"e_1_3_2_1_10_1","volume-title":"Os-kairos: Adaptive interaction for mllm-powered gui agents. arXiv preprint arXiv:2503.16465","author":"Cheng Pengzhou","year":"2025","unstructured":"Pengzhou Cheng, Zheng Wu, Zongru Wu, Aston Zhang, Zhuosheng Zhang, and Gongshen Liu. 2025. Os-kairos: Adaptive interaction for mllm-powered gui agents. arXiv preprint arXiv:2503.16465 (2025)."},{"key":"e_1_3_2_1_11_1","volume-title":"Advancing mobile gui agents: A verifier-driven approach to practical deployment. arXiv preprint arXiv:2503.15937","author":"Dai Gaole","year":"2025","unstructured":"Gaole Dai, Shiqi Jiang, Ting Cao, Yuanchun Li, Yuqing Yang, Rui Tan, Mo Li, and Lili Qiu. 2025. Advancing mobile gui agents: A verifier-driven approach to practical deployment. arXiv preprint arXiv:2503.15937 (2025)."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3126594.3126651"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.52202\/075280-1220"},{"key":"e_1_3_2_1_14_1","volume-title":"Xin Eric Wang, and Gang Wu","author":"Fan Yue","year":"2025","unstructured":"Yue Fan, Handong Zhao, Ruiyi Zhang, Yu Shen, Xin Eric Wang, and Gang Wu. 2025. Gui-bee: Align gui action grounding to novel environments via autonomous exploration. arXiv preprint arXiv:2501.13896 (2025)."},{"key":"e_1_3_2_1_15_1","volume-title":"Atomic-to-Compositional Generalization for Mobile Agents with A New Benchmark and Scheduling System. arXiv preprint arXiv:2506.08972","author":"Guo Yuan","year":"2025","unstructured":"Yuan Guo, Tingjia Miao, Zheng Wu, Pengzhou Cheng, Ming Zhou, and Zhuosheng Zhang. 2025. Atomic-to-Compositional Generalization for Mobile Agents with A New Benchmark and Scheduling System. arXiv preprint arXiv:2506.08972 (2025)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01354"},{"key":"e_1_3_2_1_17_1","volume-title":"Auitestagent: Automatic requirements oriented gui function testing. arXiv preprint arXiv:2407.09018","author":"Hu Yongxiang","year":"2024","unstructured":"Yongxiang Hu, Xuan Wang, Yingchuan Wang, Yu Zhang, Shiyu Guo, Chaoyi Chen, Xin Wang, and Yangfan Zhou. 2024. Auitestagent: Automatic requirements oriented gui function testing. arXiv preprint arXiv:2407.09018 (2024)."},{"key":"e_1_3_2_1_18_1","volume-title":"Vague, Interactive, Single-App and Unethical Instructions. arXiv preprint arXiv:2508.09057","author":"Huang Zeyu","year":"2025","unstructured":"Zeyu Huang, Juyuan Wang, Longfeng Chen, Boyi Xiao, Leng Cai, Yawen Zeng, and Jin Xu. 2025. MVISU-Bench: Benchmarking Mobile Agents for Real-World Tasks by Multi-App, Vague, Interactive, Single-App and Unethical Instructions. arXiv preprint arXiv:2508.09057 (2025)."},{"key":"e_1_3_2_1_19_1","volume-title":"Joey Tianyi Zhou, and Chi Zhang","author":"Jiang Wenjia","year":"2025","unstructured":"Wenjia Jiang, Yangyang Zhuang, Chenxi Song, Xu Yang, Joey Tianyi Zhou, and Chi Zhang. 2025. Appagentx: Evolving gui agents as proficient smartphone users. arXiv preprint arXiv:2503.02268 (2025)."},{"key":"e_1_3_2_1_20_1","volume-title":"Autogui: Scaling gui grounding with automatic functionality annotations from llms. arXiv preprint arXiv:2502.01977","author":"Li Hongxin","year":"2025","unstructured":"Hongxin Li, Jingfan Chen, Jingran Su, Yuntao Chen, Qing Li, and Zhaoxiang Zhang. 2025a. Autogui: Scaling gui grounding with automatic functionality annotations from llms. arXiv preprint arXiv:2502.01977 (2025)."},{"key":"e_1_3_2_1_21_1","volume-title":"MobileUse: A GUI Agent with Hierarchical Reflection for Autonomous Mobile Operation. arXiv preprint arXiv:2507.16853","author":"Li Ning","year":"2025","unstructured":"Ning Li, Xiangmou Qu, Jiamu Zhou, Jun Wang, Muning Wen, Kounianhua Du, Xingyu Lou, Qiuying Peng, and Weinan Zhang. 2025b. MobileUse: A GUI Agent with Hierarchical Reflection for Autonomous Mobile Operation. arXiv preprint arXiv:2507.16853 (2025)."},{"key":"e_1_3_2_1_22_1","volume-title":"On the effects of data scale on computer control agents. arXiv e-prints","author":"Li Wei","year":"2024","unstructured":"Wei Li, William Bishop, Alice Li, Chris Rawles, Folawiyo Campbell-Ajala, Divya Tyamagundlu, and Oriana Riva. 2024a. On the effects of data scale on computer control agents. arXiv e-prints (2024), arXiv-2406."},{"key":"e_1_3_2_1_23_1","volume-title":"Widget captioning: Generating natural language description for mobile user interface elements. arXiv preprint arXiv:2010.04295","author":"Li Yang","year":"2020","unstructured":"Yang Li, Gang Li, Luheng He, Jingjie Zheng, Hong Li, and Zhiwei Guan. 2020. Widget captioning: Generating natural language description for mobile user interface elements. arXiv preprint arXiv:2010.04295 (2020)."},{"key":"e_1_3_2_1_24_1","volume-title":"Omnibench: Towards the future of universal omni-language models. arXiv preprint arXiv:2409.15272","author":"Li Yizhi","year":"2024","unstructured":"Yizhi Li, Ge Zhang, Yinghao Ma, Ruibin Yuan, Kang Zhu, Hangyu Guo, Yiming Liang, Jiaheng Liu, Zekun Wang, Jian Yang, et al., 2024b. Omnibench: Towards the future of universal omni-language models. arXiv preprint arXiv:2409.15272 (2024)."},{"key":"e_1_3_2_1_25_1","volume-title":"Learnact: Few-shot mobile gui agent with a unified demonstration benchmark. arXiv preprint arXiv:2504.13805","author":"Liu Guangyi","year":"2025","unstructured":"Guangyi Liu, Pengxiang Zhao, Liang Liu, Zhiming Chen, Yuxiang Chai, Shuai Ren, Hao Wang, Shibo He, and Wenchao Meng. 2025b. Learnact: Few-shot mobile gui agent with a unified demonstration benchmark. arXiv preprint arXiv:2504.13805 (2025)."},{"key":"e_1_3_2_1_26_1","unstructured":"Shunyu Liu Minghao Liu Huichi Zhou Zhenyu Cui Yang Zhou Yuhao Zhou Wendong Fan Ge Zhang Jiajun Shi Weihao Xuan et al. 2025a. VeriGUI: Verifiable long-chain GUI dataset. arXiv preprint arXiv:2508.04026 (2025)."},{"key":"e_1_3_2_1_27_1","volume-title":"Gui odyssey: A comprehensive dataset for cross-app gui navigation on mobile devices. arXiv preprint arXiv:2406.08451","author":"Lu Quanfeng","year":"2024","unstructured":"Quanfeng Lu, Wenqi Shao, Zitao Liu, Fanqing Meng, Boxuan Li, Botong Chen, Siyuan Huang, Kaipeng Zhang, Yu Qiao, and Ping Luo. 2024. Gui odyssey: A comprehensive dataset for cross-app gui navigation on mobile devices. arXiv preprint arXiv:2406.08451 (2024)."},{"key":"e_1_3_2_1_28_1","volume-title":"Caution for the environment: Multimodal agents are susceptible to environmental distractions. arXiv preprint arXiv:2408.02544","author":"Ma Xinbei","year":"2024","unstructured":"Xinbei Ma, Yiting Wang, Yao Yao, Tongxin Yuan, Aston Zhang, Zhuosheng Zhang, and Hai Zhao. 2024. Caution for the environment: Multimodal agents are susceptible to environmental distractions. arXiv preprint arXiv:2408.02544 (2024)."},{"key":"e_1_3_2_1_29_1","unstructured":"OpenAI. 2024. GPT-4o System Card. https:\/\/cdn.openai.com\/gpt-4o-system-card.pdf."},{"key":"e_1_3_2_1_30_1","volume-title":"Ui-tars: Pioneering automated gui interaction with native agents. arXiv preprint arXiv:2501.12326","author":"Qin Yujia","year":"2025","unstructured":"Yujia Qin, Yining Ye, Junjie Fang, Haoming Wang, Shihao Liang, Shizuo Tian, Junda Zhang, Jiahao Li, Yunxin Li, Shijue Huang, et al., 2025. Ui-tars: Pioneering automated gui interaction with native agents. arXiv preprint arXiv:2501.12326 (2025)."},{"key":"e_1_3_2_1_31_1","volume-title":"Androidworld: A dynamic benchmarking environment for autonomous agents. arXiv preprint arXiv:2405.14573","author":"Rawles Christopher","year":"2024","unstructured":"Christopher Rawles, Sarah Clinckemaillie, Yifan Chang, Jonathan Waltz, Gabrielle Lau, Marybeth Fair, Alice Li, William Bishop, Wei Li, Folawiyo Campbell-Ajala, et al., 2024. Androidworld: A dynamic benchmarking environment for autonomous agents. arXiv preprint arXiv:2405.14573 (2024)."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2609"},{"key":"e_1_3_2_1_33_1","volume-title":"Meta-gui: Towards multi-modal conversational agents on mobile gui. arXiv preprint arXiv:2205.11029","author":"Sun Liangtai","year":"2022","unstructured":"Liangtai Sun, Xingyu Chen, Lu Chen, Tianle Dai, Zichen Zhu, and Kai Yu. 2022. Meta-gui: Towards multi-modal conversational agents on mobile gui. arXiv preprint arXiv:2205.11029 (2022)."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01814"},{"key":"e_1_3_2_1_35_1","first-page":"4","article-title":"An innovative technique for web text watermarking (AITW)","volume":"25","author":"Ahvanooey Milad Taleby","year":"2016","unstructured":"Milad Taleby Ahvanooey, Hassan Dana Mazraeh, and Seyed Hashem Tabasi. 2016. An innovative technique for web text watermarking (AITW). Information Security Journal: A Global Perspective, Vol. 25, 4-6 (2016), 191-196.","journal-title":"Information Security Journal: A Global Perspective"},{"key":"e_1_3_2_1_36_1","unstructured":"Qwen Team. 2023. Qwen-VL MAX. https:\/\/github.com\/QwenLM\/Qwen-VL"},{"key":"e_1_3_2_1_37_1","unstructured":"Qwen Team. 2025. Qwen3-VL. https:\/\/github.com\/QwenLM\/Qwen3-VL"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3472749.3474765"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.52202\/079017-0088"},{"key":"e_1_3_2_1_40_1","volume-title":"Mobile-agent: Autonomous multi-modal mobile device agent with visual perception. arXiv preprint arXiv:2401.16158","author":"Wang Junyang","year":"2024","unstructured":"Junyang Wang, Haiyang Xu, Jiabo Ye, Ming Yan, Weizhou Shen, Ji Zhang, Fei Huang, and Jitao Sang. 2024c. Mobile-agent: Autonomous multi-modal mobile device agent with visual perception. arXiv preprint arXiv:2401.16158 (2024)."},{"key":"e_1_3_2_1_41_1","volume-title":"Mobileagentbench: An efficient and user-friendly benchmark for mobile llm agents. arXiv preprint arXiv:2406.08184","author":"Wang Luyuan","year":"2024","unstructured":"Luyuan Wang, Yongyu Deng, Yiwei Zha, Guodong Mao, Qinmin Wang, Tianchen Min, Wei Chen, and Shoufa Chen. 2024a. Mobileagentbench: An efficient and user-friendly benchmark for mobile llm agents. arXiv preprint arXiv:2406.08184 (2024)."},{"key":"e_1_3_2_1_42_1","volume-title":"Jin Xu, Victor R\u00fchle, and Saravan Rajmohan.","author":"Wang Weixuan","year":"2025","unstructured":"Weixuan Wang, Dongge Han, Daniel Madrigal Diaz, Jin Xu, Victor R\u00fchle, and Saravan Rajmohan. 2025a. OdysseyBench: Evaluating LLM Agents on Long-Horizon Complex Office Application Workflows. arXiv preprint arXiv:2508.09124 (2025)."},{"key":"e_1_3_2_1_43_1","volume-title":"Mobile-agent-e: Self-evolving mobile assistant for complex tasks. arXiv preprint arXiv:2501.11733","author":"Wang Zhenhailong","year":"2025","unstructured":"Zhenhailong Wang, Haiyang Xu, Junyang Wang, Xi Zhang, Ming Yan, Ji Zhang, Fei Huang, and Heng Ji. 2025b. Mobile-agent-e: Self-evolving mobile assistant for complex tasks. arXiv preprint arXiv:2501.11733 (2025)."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3636534.3649379"},{"key":"e_1_3_2_1_45_1","volume-title":"Mobilevlm: A vision-language model for better intra-and inter-ui understanding. arXiv preprint arXiv:2409.14818","author":"Wu Qinzhuo","year":"2024","unstructured":"Qinzhuo Wu, Weikai Xu, Wei Liu, Tao Tan, Jianfeng Liu, Ang Li, Jian Luan, Bin Wang, and Shuo Shang. 2024b. Mobilevlm: A vision-language model for better intra-and inter-ui understanding. arXiv preprint arXiv:2409.14818 (2024)."},{"key":"e_1_3_2_1_46_1","volume-title":"Quick on the Uptake: Eliciting Implicit Intents from Human Demonstrations for Personalized Mobile-Use Agents. arXiv preprint arXiv:2508.08645","author":"Wu Zheng","year":"2025","unstructured":"Zheng Wu, Heyuan Huang, Yanjia Yang, Yuanyi Song, Xingyu Lou, Weiwen Liu, Weinan Zhang, Jun Wang, and Zhuosheng Zhang. 2025. Quick on the Uptake: Eliciting Implicit Intents from Human Demonstrations for Personalized Mobile-Use Agents. arXiv preprint arXiv:2508.08645 (2025)."},{"key":"e_1_3_2_1_47_1","volume-title":"Paul Pu Liang, et al","author":"Wu Zhiyong","year":"2024","unstructured":"Zhiyong Wu, Zhenyu Wu, Fangzhi Xu, Yian Wang, Qiushi Sun, Chengyou Jia, Kanzhi Cheng, Zichen Ding, Liheng Chen, Paul Pu Liang, et al., 2024a. Os-atlas: A foundation action model for generalist gui agents. arXiv preprint arXiv:2410.23218 (2024)."},{"key":"e_1_3_2_1_48_1","volume-title":"Mobile-Bench-v2: A More Realistic and Comprehensive Benchmark for VLM-based Mobile Agents. arXiv preprint arXiv:2505.11891","author":"Xu Weikai","year":"2025","unstructured":"Weikai Xu, Zhizheng Jiang, Yuxuan Liu, Pengzhi Gao, Wei Liu, Jian Luan, Yuanchun Li, Yunxin Liu, Bin Wang, and Bo An. 2025. Mobile-Bench-v2: A More Realistic and Comprehensive Benchmark for VLM-based Mobile Agents. arXiv preprint arXiv:2505.11891 (2025)."},{"key":"e_1_3_2_1_49_1","volume-title":"Androidlab: Training and systematic benchmarking of android autonomous agents. arXiv preprint arXiv:2410.24024","author":"Xu Yifan","year":"2024","unstructured":"Yifan Xu, Xiao Liu, Xueqiao Sun, Siyi Cheng, Hao Yu, Hanyu Lai, Shudan Zhang, Dan Zhang, Jie Tang, and Yuxiao Dong. 2024. Androidlab: Training and systematic benchmarking of android autonomous agents. arXiv preprint arXiv:2410.24024 (2024)."},{"key":"e_1_3_2_1_50_1","unstructured":"Jiabo Ye Xi Zhang Haiyang Xu Haowei Liu Junyang Wang Zhaoqing Zhu Ziwei Zheng Feiyu Gao Junjie Cao Zhengxi Lu et al. 2025b. Mobile-agent-v3: Foundamental agents for gui automation. arXiv preprint arXiv:2508.15144 (2025)."},{"key":"e_1_3_2_1_51_1","volume-title":"Realwebassist: A benchmark for long-horizon web assistance with real-world users. arXiv preprint arXiv:2504.10445","author":"Ye Suyu","year":"2025","unstructured":"Suyu Ye, Haojun Shi, Darren Shih, Hyokun Yun, Tanya Roosta, and Tianmin Shu. 2025a. Realwebassist: A benchmark for long-horizon web assistance with real-world users. arXiv preprint arXiv:2504.10445 (2025)."},{"key":"e_1_3_2_1_52_1","unstructured":"Chaoyun Zhang Shilin He Jiaxu Qian Bowen Li Liqun Li Si Qin Yu Kang Minghua Ma Guyue Liu Qingwei Lin et al. 2024a. Large language model-brained gui agents: A survey. arXiv preprint arXiv:2411.18279 (2024)."},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1145\/3706598.3713600"},{"key":"e_1_3_2_1_54_1","volume-title":"Mobile-env: Building qualified evaluation benchmarks for llm-gui interaction. arXiv preprint arXiv:2305.08144","author":"Zhang Danyang","year":"2023","unstructured":"Danyang Zhang, Zhennan Shen, Rui Xie, Situo Zhang, Tianbao Xie, Zihan Zhao, Siyuan Chen, Lu Chen, Hongshen Xu, Ruisheng Cao, et al., 2023. Mobile-env: Building qualified evaluation benchmarks for llm-gui interaction. arXiv preprint arXiv:2305.08144 (2023)."},{"key":"e_1_3_2_1_55_1","volume-title":"Android in the zoo: Chain-of-action-thought for gui agents. arXiv preprint arXiv:2403.02713","author":"Zhang Jiwen","year":"2024","unstructured":"Jiwen Zhang, Jihao Wu, Yihua Teng, Minghui Liao, Nuo Xu, Xiao Xiao, Zhongyu Wei, and Duyu Tang. 2024b. Android in the zoo: Chain-of-action-thought for gui agents. arXiv preprint arXiv:2403.02713 (2024)."},{"key":"e_1_3_2_1_56_1","volume-title":"Psysafe: A comprehensive framework for psychological-based attack, defense, and evaluation of multi-agent system safety. arXiv preprint arXiv:2401.11880","author":"Zhang Zaibin","year":"2024","unstructured":"Zaibin Zhang, Yongting Zhang, Lijun Li, Hongzhi Gao, Lijun Wang, Huchuan Lu, Feng Zhao, Yu Qiao, and Jing Shao. 2024c. Psysafe: A comprehensive framework for psychological-based attack, defense, and evaluation of multi-agent system safety. arXiv preprint arXiv:2401.11880 (2024)."}],"event":{"name":"WWW '26: The ACM Web Conference 2026","location":"Dubai United Arab Emirates","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"]},"container-title":["Proceedings of the ACM Web Conference 2026"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3774904.3792215","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T07:44:51Z","timestamp":1783151091000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774904.3792215"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,12]]},"references-count":56,"alternative-id":["10.1145\/3774904.3792215","10.1145\/3774904"],"URL":"https:\/\/doi.org\/10.1145\/3774904.3792215","relation":{},"subject":[],"published":{"date-parts":[[2026,4,12]]},"assertion":[{"value":"2026-04-12","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}