{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T07:59:14Z","timestamp":1776931154912,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":78,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100003725","name":"National Research Foundation of Korea","doi-asserted-by":"publisher","award":["2023R1A2C200520911"],"award-info":[{"award-number":["2023R1A2C200520911"]}],"id":[{"id":"10.13039\/501100003725","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Institute of Information & Communications Technology Planning & Evaluation","award":["RS-2021-II211343"],"award-info":[{"award-number":["RS-2021-II211343"]}]},{"name":"SNU-Global Excellence Research Center"},{"name":"ICT at Seoul National University"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,13]]},"DOI":"10.1145\/3772318.3790283","type":"proceedings-article","created":{"date-parts":[[2026,4,13]],"date-time":"2026-04-13T07:34:33Z","timestamp":1776065673000},"page":"1-22","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["GhostUI: Unveiling Hidden Interactions in Mobile UI"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-2557-6055","authenticated-orcid":false,"given":"Minkyu","family":"Kweon","sequence":"first","affiliation":[{"name":"Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-1685-4027","authenticated-orcid":false,"given":"Seokhyeon","family":"Park","sequence":"additional","affiliation":[{"name":"Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3075-3981","authenticated-orcid":false,"given":"Soohyun","family":"Lee","sequence":"additional","affiliation":[{"name":"Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-3379-0449","authenticated-orcid":false,"given":"You Been","family":"Lee","sequence":"additional","affiliation":[{"name":"Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-3533-6603","authenticated-orcid":false,"given":"Jeongmin","family":"Rhee","sequence":"additional","affiliation":[{"name":"Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7734-822X","authenticated-orcid":false,"given":"Jinwook","family":"Seo","sequence":"additional","affiliation":[{"name":"Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2026,4,13]]},"reference":[{"key":"e_1_3_3_3_2_2","unstructured":"Josh Achiam Steven Adler Sandhini Agarwal Lama Ahmad Ilge Akkaya Florencia\u00a0Leoni Aleman Diogo Almeida Janko Altenschmidt Sam Altman Shyamal Anadkat et\u00a0al. 2023. Gpt-4 technical report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.08774 (2023)."},{"key":"e_1_3_3_3_3_2","unstructured":"Gilles Baechler Srinivas Sunkara Maria Wang Fedir Zubach Hassan Mansoor Vincent Etter Victor C\u0103rbune Jason Lin Jindong Chen and Abhanshu Sharma. 2024. Screenai: A vision-language model for ui and infographics understanding. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2402.04615 (2024)."},{"key":"e_1_3_3_3_4_2","unstructured":"Chongyang Bai Xiaoxue Zang Ying Xu Srinivas Sunkara Abhinav Rastogi Jindong Chen et\u00a0al. 2021. Uibert: Learning generic multimodal representations for ui understanding. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2107.13731 (2021)."},{"key":"e_1_3_3_3_5_2","doi-asserted-by":"crossref","unstructured":"Hao Bai Yifei Zhou Jiayi Pan Mert Cemri Alane Suhr Sergey Levine and Aviral Kumar. 2024. Digirl: Training in-the-wild device-control agents with autonomous reinforcement learning. Advances in Neural Information Processing Systems 37 (2024) 12461\u201312495.","DOI":"10.52202\/079017-0397"},{"key":"e_1_3_3_3_6_2","unstructured":"Shuai Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge Sibo Song Kai Dang Peng Wang Shijie Wang Jun Tang et\u00a0al. 2025. Qwen2.5-vl technical report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2502.13923 (2025)."},{"key":"e_1_3_3_3_7_2","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445762"},{"key":"e_1_3_3_3_8_2","unstructured":"Andrea Burns Deniz Arsan Sanjna Agrawal Ranjitha Kumar Kate Saenko and Bryan\u00a0A Plummer. 2021. Mobile app tasks with iterative feedback (motif): Addressing task feasibility in interactive visual environments. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2104.08560 (2021)."},{"key":"e_1_3_3_3_9_2","unstructured":"Yuxiang Chai Siyuan Huang Yazhe Niu Han Xiao Liang Liu Dingyu Zhang Peng Gao Shuai Ren and Hongsheng Li. 2024. Amex: Android multi-annotation expo dataset for mobile gui agents. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.17490 (2024)."},{"key":"e_1_3_3_3_10_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.505"},{"key":"e_1_3_3_3_11_2","volume-title":"Designing mobile apps","author":"Cuello Javier","year":"2013","unstructured":"Javier Cuello and Jos\u00e9 Vittone. 2013. Designing mobile apps. Jos\u00e9 Vittone."},{"key":"e_1_3_3_3_12_2","doi-asserted-by":"publisher","DOI":"10.1145\/3126594.3126651"},{"key":"e_1_3_3_3_13_2","unstructured":"Google\u00a0Material Design. 2024. Material Design 3. https:\/\/m3.material.io Accessed: 2025-04-09."},{"key":"e_1_3_3_3_14_2","unstructured":"Nicolai Dorka Janusz Marecki and Ammar Anwar. 2024. Training a vision language model as smartphone assistant. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2404.08755 (2024)."},{"key":"e_1_3_3_3_15_2","volume-title":"Atomic design","author":"Frost Brad","year":"2016","unstructured":"Brad Frost. 2016. Atomic design. Brad Frost Pittsburgh."},{"key":"e_1_3_3_3_16_2","unstructured":"Longxi Gao Li Zhang Shihe Wang Shangguang Wang Yuanchun Li and Mengwei Xu. 2024. Mobileviews: A large-scale mobile gui dataset. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2409.14337 (2024)."},{"key":"e_1_3_3_3_17_2","doi-asserted-by":"publisher","DOI":"10.1145\/108844.108856"},{"key":"e_1_3_3_3_18_2","doi-asserted-by":"crossref","unstructured":"Akash Ghosh Arkadeep Acharya Sriparna Saha Vinija Jain and Aman Chadha. 2024. Exploring the frontier of vision-language models: A survey of current methodologies and future directions. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2404.07214 (2024).","DOI":"10.2139\/ssrn.4783140"},{"key":"e_1_3_3_3_19_2","doi-asserted-by":"publisher","DOI":"10.4324\/9781315740218"},{"key":"e_1_3_3_3_20_2","unstructured":"Boyu Gou Ruohan Wang Boyuan Zheng Yanan Xie Cheng Chang Yiheng Shu Huan Sun and Yu Su. 2024. Navigating the digital world as humans do: Universal visual grounding for gui agents. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2410.05243 (2024)."},{"key":"e_1_3_3_3_21_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01073"},{"key":"e_1_3_3_3_22_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i7.16741"},{"key":"e_1_3_3_3_23_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01354"},{"key":"e_1_3_3_3_24_2","unstructured":"Yu-Chung Hsiao Fedir Zubach Gilles Baechler Victor Carbune Jason Lin Maria Wang Srinivas Sunkara Yun Zhu and Jindong Chen. 2022. Screenqa: Large-scale question-answer pairs over mobile app screenshots. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2209.08199 (2022)."},{"key":"e_1_3_3_3_25_2","unstructured":"Edward\u00a0J Hu Yelong Shen Phillip Wallis Zeyuan Allen-Zhu Yuanzhi Li Shean Wang Lu Wang Weizhu Chen et\u00a0al. 2022. Lora: Low-rank adaptation of large language models. ICLR 1 2 (2022) 3."},{"key":"e_1_3_3_3_26_2","unstructured":"Aaron Hurst Adam Lerer Adam\u00a0P Goucher Adam Perelman Aditya Ramesh Aidan Clark AJ Ostrow Akila Welihinda Alan Hayes Alec Radford et\u00a0al. 2024. Gpt-4o system card. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2410.21276 (2024)."},{"key":"e_1_3_3_3_27_2","unstructured":"Apple Inc.2024. Human Interface Guidelines. https:\/\/developer.apple.com\/design\/human-interface-guidelines Accessed: 2025-04-09."},{"key":"e_1_3_3_3_28_2","doi-asserted-by":"crossref","unstructured":"Hamed Jelodar Yongli Wang Chi Yuan Xia Feng Xiahui Jiang Yanchao Li and Liang Zhao. 2019. Latent Dirichlet allocation (LDA) and topic modeling: models applications a survey. Multimedia tools and applications 78 11 (2019) 15169\u201315211.","DOI":"10.1007\/s11042-018-6894-4"},{"key":"e_1_3_3_3_29_2","unstructured":"Wenjia Jiang Yangyang Zhuang Chenxi Song Xu Yang and Chi Zhang. 2025. AppAgentX: Evolving GUI Agents as Proficient Smartphone Users. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.02268 (2025)."},{"key":"e_1_3_3_3_30_2","doi-asserted-by":"crossref","unstructured":"J\u00a0Richard Landis and Gary\u00a0G Koch. 1977. The measurement of observer agreement for categorical data. biometrics (1977) 159\u2013174.","DOI":"10.2307\/2529310"},{"key":"e_1_3_3_3_31_2","unstructured":"Juyong Lee Taywon Min Minyong An Dongyoon Hahm Haeone Lee Changyeon Kim and Kimin Lee. 2024. Benchmarking Mobile Device Control Agents across Diverse Configurations. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2404.16660 (2024)."},{"key":"e_1_3_3_3_32_2","doi-asserted-by":"publisher","DOI":"10.1145\/2399016.2399024"},{"key":"e_1_3_3_3_33_2","unstructured":"Sunjae Lee Junyoung Choi Jungjae Lee Munim\u00a0Hasan Wasi Hojun Choi Steven\u00a0Y Ko Sangeun Oh and Insik Shin. 2023. Explore select derive and recall: Augmenting llm with human-like memory for mobile task automation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2312.03003 (2023)."},{"key":"e_1_3_3_3_34_2","doi-asserted-by":"publisher","DOI":"10.1145\/3406324.3410710"},{"key":"e_1_3_3_3_35_2","unstructured":"Gang Li and Yang Li. 2022. Spotlight: Mobile ui understanding using vision-language models with a focus. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2209.14927 (2022)."},{"key":"e_1_3_3_3_36_2","unstructured":"Tao Li Gang Li Jingjie Zheng Purple Wang and Yang Li. 2022. Mug: Interactive multimodal grounding on user interfaces. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2209.15099 (2022)."},{"key":"e_1_3_3_3_37_2","unstructured":"Yang Li Jiacong He Xin Zhou Yuan Zhang and Jason Baldridge. 2020. Mapping natural language instructions to mobile UI action sequences. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2005.03776 (2020)."},{"key":"e_1_3_3_3_38_2","unstructured":"Yang Li Gang Li Luheng He Jingjie Zheng Hong Li and Zhiwei Guan. 2020. Widget captioning: Generating natural language description for mobile user interface elements. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2010.04295 (2020)."},{"key":"e_1_3_3_3_39_2","unstructured":"Yang Li Gang Li Xin Zhou Mostafa Dehghani and Alexey Gritsenko. 2021. Vut: Versatile ui transformer for multi-modal multi-task user interface modeling. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2112.05692 (2021)."},{"key":"e_1_3_3_3_40_2","unstructured":"Yanda Li Chi Zhang Wanqi Yang Bin Fu Pei Cheng Xin Chen Ling Chen and Yunchao Wei. 2024. Appagent v2: Advanced agent for flexible mobile interactions. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2408.11824 (2024)."},{"key":"e_1_3_3_3_41_2","unstructured":"Zhangheng Li Keen You Haotian Zhang Di Feng Harsh Agrawal Xiujun Li Mohana Prasad\u00a0Sathya Moorthy Jeff Nichols Yinfei Yang and Zhe Gan. 2024. Ferret-ui 2: Mastering universal user interface understanding across platforms. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2410.18967 (2024)."},{"key":"e_1_3_3_3_42_2","doi-asserted-by":"publisher","DOI":"10.1145\/3491102.3501992"},{"key":"e_1_3_3_3_43_2","unstructured":"Quanfeng Lu Wenqi Shao Zitao Liu Fanqing Meng Boxuan Li Botong Chen Siyuan Huang Kaipeng Zhang Yu Qiao and Ping Luo. 2024. Gui odyssey: A comprehensive dataset for cross-app gui navigation on mobile devices. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.08451 (2024)."},{"key":"e_1_3_3_3_44_2","unstructured":"Yadong Lu Jianwei Yang Yelong Shen and Ahmed Awadallah. 2024. Omniparser for pure vision based gui agent. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2408.00203 (2024)."},{"key":"e_1_3_3_3_45_2","unstructured":"Xinbei Ma Zhuosheng Zhang and Hai Zhao. 2024. Coco-agent: A comprehensive cognitive mllm agent for smartphone gui automation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2402.11941 (2024)."},{"key":"e_1_3_3_3_46_2","doi-asserted-by":"crossref","unstructured":"Mubashar Munir and Pietro Murano. 2023. The Usability of Hidden Functional Elements in Mobile User Interfaces. (2023).","DOI":"10.5220\/0011827700003467"},{"key":"e_1_3_3_3_47_2","doi-asserted-by":"crossref","unstructured":"Erik\u00a0G Nilsson. 2009. Design patterns for user interface for mobile applications. Advances in engineering software 40 12 (2009) 1318\u20131328.","DOI":"10.1016\/j.advengsoft.2009.01.017"},{"key":"e_1_3_3_3_48_2","doi-asserted-by":"crossref","unstructured":"Donald\u00a0A Norman. 1999. Affordance conventions and design. interactions 6 3 (1999) 38\u201343.","DOI":"10.1145\/301153.301168"},{"key":"e_1_3_3_3_49_2","unstructured":"OpenAI. 2024. Fine-tuning Guide. https:\/\/platform.openai.com\/docs\/guides\/fine-tuning Accessed: 2025-04-08."},{"key":"e_1_3_3_3_50_2","unstructured":"Seokhyeon Park Wonjae Kim Young-Ho Kim and Jinwook Seo. 2023. Computational approaches for app-to-app retrieval and design consistency check. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2309.10328 (2023)."},{"key":"e_1_3_3_3_51_2","doi-asserted-by":"publisher","DOI":"10.1145\/3706598.3714213"},{"key":"e_1_3_3_3_52_2","doi-asserted-by":"crossref","unstructured":"Panupong Pasupat Tian-Shun Jiang Evan\u00a0Zheran Liu Kelvin Guu and Percy Liang. 2018. Mapping natural language commands to web elements. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1808.09132 (2018).","DOI":"10.18653\/v1\/D18-1540"},{"key":"e_1_3_3_3_53_2","unstructured":"Christopher Rawles Sarah Clinckemaillie Yifan Chang Jonathan Waltz Gabrielle Lau Marybeth Fair Alice Li William Bishop Wei Li Folawiyo Campbell-Ajala et\u00a0al. 2024. Androidworld: A dynamic benchmarking environment for autonomous agents. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2405.14573 (2024)."},{"key":"e_1_3_3_3_54_2","doi-asserted-by":"crossref","unstructured":"Christopher Rawles Alice Li Daniel Rodriguez Oriana Riva and Timothy Lillicrap. 2023. Androidinthewild: A large-scale dataset for android device control. Advances in Neural Information Processing Systems 36 (2023) 59708\u201359728.","DOI":"10.52202\/075280-2609"},{"key":"e_1_3_3_3_55_2","doi-asserted-by":"publisher","DOI":"10.1145\/3491102.3517497"},{"key":"e_1_3_3_3_56_2","doi-asserted-by":"publisher","DOI":"10.21428\/594757db.e57f0d1e"},{"key":"e_1_3_3_3_57_2","doi-asserted-by":"publisher","DOI":"10.1145\/3654777.3676386"},{"key":"e_1_3_3_3_58_2","unstructured":"Liangtai Sun Xingyu Chen Lu Chen Tianle Dai Zichen Zhu and Kai Yu. 2022. Meta-gui: Towards multi-modal conversational agents on mobile gui. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2205.11029 (2022)."},{"key":"e_1_3_3_3_59_2","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300305"},{"key":"e_1_3_3_3_60_2","unstructured":"Daniel Toyama Philippe Hamel Anita Gergely Gheorghe Comanici Amelia Glaese Zafarali Ahmed Tyler Jackson Shibl Mourad and Doina Precup. 2021. Androidenv: A reinforcement learning platform for android. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2105.13231 (2021)."},{"key":"e_1_3_3_3_61_2","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3580895"},{"key":"e_1_3_3_3_62_2","doi-asserted-by":"publisher","DOI":"10.1145\/3472749.3474765"},{"key":"e_1_3_3_3_63_2","unstructured":"Zhenhailong Wang Haiyang Xu Junyang Wang Xi Zhang Ming Yan Ji Zhang Fei Huang and Heng Ji. 2025. Mobile-Agent-E: Self-Evolving Mobile Assistant for Complex Tasks. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2501.11733 (2025)."},{"key":"e_1_3_3_3_64_2","doi-asserted-by":"publisher","DOI":"10.1145\/3636534.3649379"},{"key":"e_1_3_3_3_65_2","unstructured":"Hao Wen Hongming Wang Jiaxuan Liu and Yuanchun Li. 2023. Droidbot-gpt: Gpt-powered ui automation for android. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2304.07061 (2023)."},{"key":"e_1_3_3_3_66_2","doi-asserted-by":"crossref","unstructured":"Max Wertheimer. 1938. Gestalt theory. (1938).","DOI":"10.1037\/11496-001"},{"key":"e_1_3_3_3_67_2","doi-asserted-by":"crossref","unstructured":"Frank Wilcoxon. 1945. Individual comparisons by ranking methods. Biometrics bulletin 1 6 (1945) 80\u201383.","DOI":"10.2307\/3001968"},{"key":"e_1_3_3_3_68_2","unstructured":"Biao Wu Yanda Li Meng Fang Zirui Song Zhiwei Zhang Yunchao Wei and Ling Chen. 2024. Foundations and recent trends in multimodal mobile agents: A survey. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2411.02006 (2024)."},{"key":"e_1_3_3_3_69_2","doi-asserted-by":"publisher","DOI":"10.1145\/3586183.3606824"},{"key":"e_1_3_3_3_70_2","unstructured":"Qinzhuo Wu Weikai Xu Wei Liu Tao Tan Jianfeng Liu Ang Li Jian Luan Bin Wang and Shuo Shang. 2024. Mobilevlm: A vision-language model for better intra-and inter-ui understanding. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2409.14818 (2024)."},{"key":"e_1_3_3_3_71_2","unstructured":"Wenhao Wu Huanjin Yao Mengxi Zhang Yuxin Song Wanli Ouyang and Jingdong Wang. 2023. GPT4Vis: what can GPT-4 do for zero-shot visual recognition? arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2311.15732 (2023)."},{"key":"e_1_3_3_3_72_2","doi-asserted-by":"crossref","unstructured":"Shuhong Xiao Yunnong Chen Yaxuan Song Liuqing Chen Lingyun Sun Yankun Zhen and Yanfang Chang. 2024. UI Semantic Group Detection: Grouping UI Elements with Similar Semantics in Mobile Graphical User Interface. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2403.04984 (2024).","DOI":"10.1016\/j.displa.2024.102679"},{"key":"e_1_3_3_3_73_2","unstructured":"Yiheng Xu Zekun Wang Junli Wang Dunjie Lu Tianbao Xie Amrita Saha Doyen Sahoo Tao Yu and Caiming Xiong. 2024. Aguvis: Unified pure vision agents for autonomous gui interaction. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.04454 (2024)."},{"key":"e_1_3_3_3_74_2","first-page":"240","volume-title":"European Conference on Computer Vision","author":"You Keen","year":"2024","unstructured":"Keen You, Haotian Zhang, Eldon Schoop, Floris Weers, Amanda Swearngin, Jeffrey Nichols, Yinfei Yang, and Zhe Gan. 2024. Ferret-ui: Grounded mobile ui understanding with multimodal llms. In European Conference on Computer Vision. Springer, 240\u2013255."},{"key":"e_1_3_3_3_75_2","unstructured":"Danyang Zhang Zhennan Shen Rui Xie Situo Zhang Tianbao Xie Zihan Zhao Siyuan Chen Lu Chen Hongshen Xu Ruisheng Cao et\u00a0al. 2023. Mobile-Env: Building Qualified Evaluation Benchmarks for LLM-GUI Interaction. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2305.08144 (2023)."},{"key":"e_1_3_3_3_76_2","doi-asserted-by":"crossref","unstructured":"Jingyi Zhang Jiaxing Huang Sheng Jin and Shijian Lu. 2024. Vision-language models for vision tasks: A survey. IEEE Transactions on Pattern Analysis and Machine Intelligence (2024).","DOI":"10.1109\/TPAMI.2024.3369699"},{"key":"e_1_3_3_3_77_2","doi-asserted-by":"publisher","DOI":"10.1145\/3654777.3676382"},{"key":"e_1_3_3_3_78_2","unstructured":"Zhuosheng Zhang and Aston Zhang. 2023. You only look at screens: Multimodal chain-of-action agents. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2309.11436 (2023)."},{"key":"e_1_3_3_3_79_2","unstructured":"Zhizheng Zhang Xiaoyi Zhang Wenxuan Xie and Yan Lu. 2023. Responsible task automation: Empowering large language models as responsible task automators. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2306.01242 (2023)."}],"event":{"name":"CHI 2026: CHI Conference on Human Factors in Computing Systems","location":"Barcelona Spain","acronym":"CHI '26","sponsor":["SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["Proceedings of the 2026 CHI Conference on Human Factors in Computing Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3772318.3790283","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,17]],"date-time":"2026-04-17T09:11:57Z","timestamp":1776417117000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3772318.3790283"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,13]]},"references-count":78,"alternative-id":["10.1145\/3772318.3790283","10.1145\/3772318"],"URL":"https:\/\/doi.org\/10.1145\/3772318.3790283","relation":{},"subject":[],"published":{"date-parts":[[2026,4,13]]},"assertion":[{"value":"2026-04-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}