{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T15:44:13Z","timestamp":1782834253427,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":62,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,7,20]],"date-time":"2025-07-20T00:00:00Z","timestamp":1752969600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62122089"],"award-info":[{"award-number":["62122089"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,7,20]]},"DOI":"10.1145\/3690624.3709171","type":"proceedings-article","created":{"date-parts":[[2025,4,4]],"date-time":"2025-04-04T18:44:43Z","timestamp":1743792283000},"page":"883-893","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":6,"title":["MobileSteward: Integrating Multiple App-Oriented Agents with Self-Evolution to Automate Cross-App Instructions"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-2154-7215","authenticated-orcid":false,"given":"Yuxuan","family":"Liu","sequence":"first","affiliation":[{"name":"Gaoling School of Artificial Intelligence, Renmin University of China, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4850-6134","authenticated-orcid":false,"given":"Hongda","family":"Sun","sequence":"additional","affiliation":[{"name":"Gaoling School of Artificial Intelligence, Renmin University of China, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-4327-1920","authenticated-orcid":false,"given":"Wei","family":"Liu","sequence":"additional","affiliation":[{"name":"Xiaomi AI Lab, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2383-226X","authenticated-orcid":false,"given":"Jian","family":"Luan","sequence":"additional","affiliation":[{"name":"Xiaomi AI Lab, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0059-8458","authenticated-orcid":false,"given":"Bo","family":"Du","sequence":"additional","affiliation":[{"name":"School of Computer Science, Wuhan University, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3356-6823","authenticated-orcid":false,"given":"Rui","family":"Yan","sequence":"additional","affiliation":[{"name":"Gaoling School of Artificial Intelligence, Renmin University of China, Beijing, China and School of Computer Science, Wuhan University, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,7,20]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Screenai: A vision-language model for ui and infographics understanding. arXiv preprint arXiv:2402.04615","author":"Baechler Gilles","year":"2024","unstructured":"Gilles Baechler, Srinivas Sunkara, Maria Wang, Fedir Zubach, Hassan Mansoor, Vincent Etter, Victor C\u0103rbune, Jason Lin, Jindong Chen, and Abhanshu Sharma. 2024. Screenai: A vision-language model for ui and infographics understanding. arXiv preprint arXiv:2402.04615 (2024)."},{"key":"e_1_3_2_2_2_1","volume-title":"Uibert: Learning generic multimodal representations for ui understanding. arXiv preprint arXiv:2107.13731","author":"Bai Chongyang","year":"2021","unstructured":"Chongyang Bai, Xiaoxue Zang, Ying Xu, Srinivas Sunkara, Abhinav Rastogi, Jindong Chen, et al. 2021. Uibert: Learning generic multimodal representations for ui understanding. arXiv preprint arXiv:2107.13731 (2021)."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20074-8_18"},{"key":"e_1_3_2_2_4_1","volume-title":"Interactive mobile app navigation with uncertain or under-specified natural language commands. arXiv preprint arXiv:2202.02312","author":"Burns Andrea","year":"2022","unstructured":"Andrea Burns, Deniz Arsan, Sanjna Agrawal, Ranjitha Kumar, Kate Saenko, and Bryan A Plummer. 2022. Interactive mobile app navigation with uncertain or under-specified natural language commands. arXiv preprint arXiv:2202.02312 (2022)."},{"key":"e_1_3_2_2_5_1","volume-title":"Chateval: Towards better llm-based evaluators through multi-agent debate. arXiv preprint arXiv:2308.07201","author":"Chan Chi-Min","year":"2023","unstructured":"Chi-Min Chan, Weize Chen, Yusheng Su, Jianxuan Yu, Wei Xue, Shanghang Zhang, Jie Fu, and Zhiyuan Liu. 2023. Chateval: Towards better llm-based evaluators through multi-agent debate. arXiv preprint arXiv:2308.07201 (2023)."},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3491102.3502073"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"crossref","unstructured":"Shihan Deng Weikai Xu Hongda Sun Wei Liu Tao Tan Jianfeng Liu Ang Li Jian Luan Bin Wang Rui Yan et al. 2024. Mobile-Bench: An Evaluation Benchmark for LLM-based Mobile Agents. arXiv preprint arXiv:2407.00993 (2024).","DOI":"10.18653\/v1\/2024.acl-long.478"},{"key":"e_1_3_2_2_8_1","volume-title":"The Workshop on Generative Models for Decision Making of Eleventh International Conference on Learning Representations.","author":"Dorka Nicolai","year":"2024","unstructured":"Nicolai Dorka, Janusz Marecki, and Ammar Anwar. 2024. Training a Vision Language Model as Smartphone Assistant. In The Workshop on Generative Models for Decision Making of Eleventh International Conference on Learning Representations."},{"key":"e_1_3_2_2_9_1","volume-title":"Improving factuality and reasoning in language models through multiagent debate. arXiv preprint arXiv:2305.14325","author":"Du Yilun","year":"2023","unstructured":"Yilun Du, Shuang Li, Antonio Torralba, Joshua B Tenenbaum, and Igor Mordatch. 2023. Improving factuality and reasoning in language models through multiagent debate. arXiv preprint arXiv:2305.14325 (2023)."},{"key":"e_1_3_2_2_10_1","volume-title":"Openagi: When llm meets domain experts. Advances in Neural Information Processing Systems 36","author":"Ge Yingqiang","year":"2024","unstructured":"Yingqiang Ge, Wenyue Hua, Kai Mei, Juntao Tan, Shuyuan Xu, Zelong Li, Yongfeng Zhang, et al. 2024. Openagi: When llm meets domain experts. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2024\/890"},{"key":"e_1_3_2_2_12_1","volume-title":"MetaGPT: Meta Programming for A Multi-Agent Collaborative Framework. In The Twelfth International Conference on Learning Representations.","author":"Hong Sirui","year":"2024","unstructured":"Sirui Hong, Mingchen Zhuge, Jonathan Chen, Xiawu Zheng, Yuheng Cheng, Jinlin Wang, Ceyao Zhang, Zili Wang, Steven Ka Shing Yau, Zijuan Lin, et al. 2024. MetaGPT: Meta Programming for A Multi-Agent Collaborative Framework. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01354"},{"key":"e_1_3_2_2_14_1","volume-title":"Man Ho Lam, Tian Liang, Wenxuan Wang, Youliang Yuan, Wenxiang Jiao, Xing Wang, Zhaopeng Tu, and Michael R Lyu.","year":"2024","unstructured":"Jen-tse Huang, Eric John Li, Man Ho Lam, Tian Liang, Wenxuan Wang, Youliang Yuan, Wenxiang Jiao, Xing Wang, Zhaopeng Tu, and Michael R Lyu. 2024. How Far Are We on the Decision-Making of LLMs? Evaluating LLMs' Gaming Ability in Multi-Agent Environments. CoRR (2024)."},{"key":"e_1_3_2_2_15_1","volume-title":"Guangyu Robert Yang, and Andrew Ahn","author":"Kaiya Zhao","year":"2023","unstructured":"Zhao Kaiya, Michelangelo Naim, Jovana Kondic, Manuel Cortes, Jiaxin Ge, Shuying Luo, Guangyu Robert Yang, and Andrew Ahn. 2023. Lyfe agents: Generative agents for low-cost real-time social interactions. arXiv preprint arXiv:2310.02172 (2023)."},{"key":"e_1_3_2_2_16_1","volume-title":"Manling Li, and Heng Ji.","author":"Kim Kyungha","year":"2024","unstructured":"Kyungha Kim, Sangyun Lee, Kung-Hsiang Huang, Hou Pong Chan, Manling Li, and Heng Ji. 2024. Can LLMs Produce Faithful Explanations For Fact-checking? Towards Faithful Explainable Fact-Checking via Multi-Agent Debate. arXiv preprint arXiv:2402.07401 (2024)."},{"key":"e_1_3_2_2_17_1","volume-title":"Augmenting LLM with Human-like Memory for Mobile Task Automation. arXiv preprint arXiv:2312.03003","author":"Lee Sunjae","year":"2023","unstructured":"Sunjae Lee, Junyoung Choi, Jungjae Lee, Hojun Choi, Steven Y Ko, Sangeun Oh, and Insik Shin. 2023. Explore, Select, Derive, and Recall: Augmenting LLM with Human-like Memory for Mobile Task Automation. arXiv preprint arXiv:2312.03003 (2023)."},{"key":"e_1_3_2_2_18_1","first-page":"51991","article-title":"Camel: Communicative agents for\" mind\" exploration of large language model society","volume":"36","author":"Li Guohao","year":"2023","unstructured":"Guohao Li, Hasan Hammoud, Hani Itani, Dmitrii Khizbullin, and Bernard Ghanem. 2023. Camel: Communicative agents for\" mind\" exploration of large language model society. Advances in Neural Information Processing Systems 36 (2023), 51991--52008.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_19_1","volume-title":"The Eleventh International Conference on Learning Representations.","author":"Li Gang","year":"2023","unstructured":"Gang Li and Yang Li. 2023. Spotlight: Mobile UI Understanding using Vision- Language Models with a Focus. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_2_2_20_1","volume-title":"Proceedings of the 2021 CHI Conference on Human Factors in Computing Systems. 1--15","author":"Jia-Jun Li Toby","year":"2021","unstructured":"Toby Jia-Jun Li, Lindsay Popowski, Tom Mitchell, and Brad A Myers. 2021. Screen2vec: Semantic embedding of gui screens and gui components. In Proceedings of the 2021 CHI Conference on Human Factors in Computing Systems. 1--15."},{"key":"e_1_3_2_2_21_1","volume-title":"Learning ui navigation through demonstrations composed of macro actions. arXiv preprint arXiv:2110.08653","author":"Wei Li.","year":"2021","unstructured":"Wei Li. 2021. Learning ui navigation through demonstrations composed of macro actions. arXiv preprint arXiv:2110.08653 (2021)."},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.729"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.443"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/62084.62113"},{"key":"e_1_3_2_2_25_1","volume-title":"From Skepticism to Acceptance: Simulating the Attitude Dynamics Toward Fake News. arXiv preprint arXiv:2403.09498","author":"Liu Yuhan","year":"2024","unstructured":"Yuhan Liu, Xiuying Chen, Xiaoqing Zhang, Xing Gao, Ji Zhang, and Rui Yan. 2024. From Skepticism to Acceptance: Simulating the Attitude Dynamics Toward Fake News. arXiv preprint arXiv:2403.09498 (2024)."},{"key":"e_1_3_2_2_26_1","volume-title":"From a tiny slip to a giant leap: An llm-based simulation for fake news evolution. arXiv preprint arXiv:2410.19064","author":"Liu Yuhan","year":"2024","unstructured":"Yuhan Liu, Zirui Song, Xiaoqing Zhang, Xiuying Chen, and Rui Yan. 2024. From a tiny slip to a giant leap: An llm-based simulation for fake news evolution. arXiv preprint arXiv:2410.19064 (2024)."},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i1.32034"},{"key":"e_1_3_2_2_28_1","volume-title":"Comprehensive Cognitive LLM Agent for Smartphone GUI Automation. arXiv preprint arXiv:2402.11941","author":"Ma Xinbei","year":"2024","unstructured":"Xinbei Ma, Zhuosheng Zhang, and Hai Zhao. 2024. Comprehensive Cognitive LLM Agent for Smartphone GUI Automation. arXiv preprint arXiv:2402.11941 (2024)."},{"key":"e_1_3_2_2_29_1","volume-title":"LLMs with Personalities in Multiissue Negotiation Games. arXiv preprint arXiv:2405.05248","author":"Noh Sean","year":"2024","unstructured":"Sean Noh and Ho-Chun Herbert Chang. 2024. LLMs with Personalities in Multiissue Negotiation Games. arXiv preprint arXiv:2405.05248 (2024)."},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3586183.3606763"},{"key":"e_1_3_2_2_31_1","volume-title":"Communicative agents for software development. arXiv preprint arXiv:2307.07924 6","author":"Qian Chen","year":"2023","unstructured":"Chen Qian and Xin Cong. 2023. Communicative agents for software development. arXiv preprint arXiv:2307.07924 6 (2023)."},{"key":"e_1_3_2_2_32_1","volume-title":"Androidinthewild: A large-scale dataset for android device control. Advances in Neural Information Processing Systems 36","author":"Rawles Christopher","year":"2024","unstructured":"Christopher Rawles, Alice Li, Daniel Rodriguez, Oriana Riva, and Timothy Lillicrap. 2024. Androidinthewild: A large-scale dataset for android device control. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"crossref","unstructured":"Stephen Robertson Hugo Zaragoza et al. 2009. The probabilistic relevance framework: BM25 and beyond. Foundations and Trends\u00ae in Information Retrieval 3 4 (2009) 333--389.","DOI":"10.1561\/1500000019"},{"key":"e_1_3_2_2_34_1","volume-title":"Hugginggpt: Solving ai tasks with chatgpt and its friends in hugging face. Advances in Neural Information Processing Systems 36","author":"Shen Yongliang","year":"2024","unstructured":"Yongliang Shen, Kaitao Song, Xu Tan, Dongsheng Li, Weiming Lu, and Yueting Zhuang. 2024. Hugginggpt: Solving ai tasks with chatgpt and its friends in hugging face. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_2_2_35_1","volume-title":"Facilitating Multi-Role and Multi-Behavior Collaboration of Large Language Models for Online Job Seeking and Recruiting. arXiv preprint arXiv:2405.18113","author":"Sun Hongda","year":"2024","unstructured":"Hongda Sun, Hongzhan Lin, Haiyu Yan, Chen Zhu, Yang Song, Xin Gao, Shuo Shang, and Rui Yan. 2024. Facilitating Multi-Role and Multi-Behavior Collaboration of Large Language Models for Online Job Seeking and Recruiting. arXiv preprint arXiv:2405.18113 (2024)."},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589334.3645670"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.531"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.449"},{"key":"e_1_3_2_2_39_1","unstructured":"AutoGPT Team. 2023. AutoGPT. https:\/\/github.com\/Significant-Gravitas\/AutoGPT"},{"key":"e_1_3_2_2_40_1","volume-title":"Ugif: Ui grounded instruction following. arXiv preprint arXiv:2211.07615","author":"Venkatesh Sagar Gubbi","year":"2022","unstructured":"Sagar Gubbi Venkatesh, Partha Talukdar, and Srini Narayanan. 2022. Ugif: Ui grounded instruction following. arXiv preprint arXiv:2211.07615 (2022)."},{"key":"e_1_3_2_2_41_1","volume-title":"Proceedings of the 2023 CHI Conference on Human Factors in Computing Systems. 1--17","author":"Li Gang","year":"2023","unstructured":"BryanWang, Gang Li, and Yang Li. 2023. Enabling conversational interaction with mobile ui using large language models. In Proceedings of the 2023 CHI Conference on Human Factors in Computing Systems. 1--17."},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3472749.3474765"},{"key":"e_1_3_2_2_43_1","volume-title":"Mobile-Agent-v2: Mobile Device Operation Assistant with Effective Navigation via Multi-Agent Collaboration. arXiv preprint arXiv:2406.01014","author":"Wang Junyang","year":"2024","unstructured":"Junyang Wang, Haiyang Xu, Haitao Jia, Xi Zhang, Ming Yan, Weizhou Shen, Ji Zhang, Fei Huang, and Jitao Sang. 2024. Mobile-Agent-v2: Mobile Device Operation Assistant with Effective Navigation via Multi-Agent Collaboration. arXiv preprint arXiv:2406.01014 (2024)."},{"key":"e_1_3_2_2_44_1","volume-title":"Mobile-agent: Autonomous multi-modal mobile device agent with visual perception. arXiv preprint arXiv:2401.16158","author":"Wang Junyang","year":"2024","unstructured":"Junyang Wang, Haiyang Xu, Jiabo Ye, Ming Yan, Weizhou Shen, Ji Zhang, Fei Huang, and Jitao Sang. 2024. Mobile-agent: Autonomous multi-modal mobile device agent with visual perception. arXiv preprint arXiv:2401.16158 (2024)."},{"key":"e_1_3_2_2_45_1","volume-title":"MobileAgentBench: An Efficient and User-Friendly Benchmark for Mobile LLM Agents. arXiv preprint arXiv:2406.08184","author":"Wang Luyuan","year":"2024","unstructured":"Luyuan Wang, Yongyu Deng, Yiwei Zha, Guodong Mao, Qinmin Wang, Tianchen Min, Wei Chen, and Shoufa Chen. 2024. MobileAgentBench: An Efficient and User-Friendly Benchmark for Mobile LLM Agents. arXiv preprint arXiv:2406.08184 (2024)."},{"key":"e_1_3_2_2_46_1","volume-title":"Multi-Agent Collaboration Framework for Recommender Systems. arXiv preprint arXiv:2402.15235","author":"Wang Zhefan","year":"2024","unstructured":"Zhefan Wang, Yuanqing Yu, Wendi Zheng, Weizhi Ma, and Min Zhang. 2024. Multi-Agent Collaboration Framework for Recommender Systems. arXiv preprint arXiv:2402.15235 (2024)."},{"key":"e_1_3_2_2_47_1","volume-title":"Shiqi Jiang, Yunhao Liu, Yaqin Zhang, and Yunxin Liu.","author":"Wen Hao","year":"2023","unstructured":"Hao Wen, Yuanchun Li, Guohong Liu, Shanhui Zhao, Tao Yu, Toby Jia-Jun Li, Shiqi Jiang, Yunhao Liu, Yaqin Zhang, and Yunxin Liu. 2023. Empowering llm to use smartphone for intelligent task automation. arXiv preprint arXiv:2308.15272 (2023)."},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/3636534.3649379"},{"key":"e_1_3_2_2_49_1","volume-title":"Droidbot-gpt: Gpt-powered ui automation for android. arXiv preprint arXiv:2304.07061","author":"Wen Hao","year":"2023","unstructured":"Hao Wen, Hongming Wang, Jiaxuan Liu, and Yuanchun Li. 2023. Droidbot-gpt: Gpt-powered ui automation for android. arXiv preprint arXiv:2304.07061 (2023)."},{"key":"e_1_3_2_2_50_1","volume-title":"Autogen: Enabling next-gen llm applications via multi-agent conversation framework. arXiv preprint arXiv:2308.08155","author":"Wu Qingyun","year":"2023","unstructured":"Qingyun Wu, Gagan Bansal, Jieyu Zhang, Yiran Wu, Shaokun Zhang, Erkang Zhu, Beibin Li, Li Jiang, Xiaoyun Zhang, and ChiWang. 2023. Autogen: Enabling next-gen llm applications via multi-agent conversation framework. arXiv preprint arXiv:2308.08155 (2023)."},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.599"},{"key":"e_1_3_2_2_52_1","doi-asserted-by":"publisher","DOI":"10.1145\/3637528.3671650"},{"key":"e_1_3_2_2_53_1","unstructured":"An Yan Zhengyuan Yang Wanrong Zhu Kevin Lin Linjie Li Jianfeng Wang Jianwei Yang Yiwu Zhong Julian McAuley Jianfeng Gao et al. 2023. Gpt-4v in wonderland: Large multimodal models for zero-shot smartphone gui navigation. arXiv preprint arXiv:2311.07562 (2023)."},{"key":"e_1_3_2_2_54_1","volume-title":"Set-of-mark prompting unleashes extraordinary visual grounding in gpt-4v. arXiv preprint arXiv:2310.11441","author":"Yang Jianwei","year":"2023","unstructured":"Jianwei Yang, Hao Zhang, Feng Li, Xueyan Zou, Chunyuan Li, and Jianfeng Gao. 2023. Set-of-mark prompting unleashes extraordinary visual grounding in gpt-4v. arXiv preprint arXiv:2310.11441 (2023)."},{"key":"e_1_3_2_2_55_1","volume-title":"Appagent: Multimodal agents as smartphone users. arXiv preprint arXiv:2312.13771","author":"Yang Zhao","year":"2023","unstructured":"Zhao Yang, Jiaxuan Liu, Yucheng Han, Xin Chen, Zebiao Huang, Bin Fu, and Gang Yu. 2023. Appagent: Multimodal agents as smartphone users. arXiv preprint arXiv:2312.13771 (2023)."},{"key":"e_1_3_2_2_56_1","volume-title":"You only look at screens: Multimodal chain-of-action agents. arXiv preprint arXiv:2309.11436","author":"Zhan Zhuosheng","year":"2023","unstructured":"Zhuosheng Zhan and Aston Zhang. 2023. You only look at screens: Multimodal chain-of-action agents. arXiv preprint arXiv:2309.11436 (2023)."},{"key":"e_1_3_2_2_57_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657844"},{"key":"e_1_3_2_2_58_1","volume-title":"Android in the zoo: Chain-of-action-thought for gui agents. arXiv preprint arXiv:2403.02713","author":"Zhang Jiwen","year":"2024","unstructured":"Jiwen Zhang, Jihao Wu, Yihua Teng, Minghui Liao, Nuo Xu, Xiao Xiao, Zhongyu Wei, and Duyu Tang. 2024. Android in the zoo: Chain-of-action-thought for gui agents. arXiv preprint arXiv:2403.02713 (2024)."},{"key":"e_1_3_2_2_59_1","volume-title":"Exploring collaboration mechanisms for llm agents: A social psychology view. arXiv preprint arXiv:2310.02124","author":"Zhang Jintian","year":"2023","unstructured":"Jintian Zhang, Xin Xu, and Shumin Deng. 2023. Exploring collaboration mechanisms for llm agents: A social psychology view. arXiv preprint arXiv:2310.02124 (2023)."},{"key":"e_1_3_2_2_60_1","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445186"},{"key":"e_1_3_2_2_61_1","volume-title":"Forty-first International Conference on Machine Learning.","author":"Zheng Boyuan","year":"2024","unstructured":"Boyuan Zheng, Boyu Gou, Jihyung Kil, Huan Sun, and Yu Su. 2024. GPT-4V (ision) is a Generalist Web Agent, if Grounded. In Forty-first International Conference on Machine Learning."},{"key":"e_1_3_2_2_62_1","volume-title":"Enhancing LLM-based Multi-Agent Communication and Interaction in Murder Mystery Games. arXiv preprint arXiv:2404.17662","author":"Zhu Qinglin","year":"2024","unstructured":"Qinglin Zhu, Runcong Zhao, Jinhua Du, Lin Gui, and Yulan He. 2024. PLAYER*: Enhancing LLM-based Multi-Agent Communication and Interaction in Murder Mystery Games. arXiv preprint arXiv:2404.17662 (2024)."}],"event":{"name":"KDD '25: The 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Toronto ON Canada","acronym":"KDD '25","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data"]},"container-title":["Proceedings of the 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.1"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3690624.3709171","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3690624.3709171","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,16]],"date-time":"2025-08-16T15:47:00Z","timestamp":1755359220000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3690624.3709171"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,20]]},"references-count":62,"alternative-id":["10.1145\/3690624.3709171","10.1145\/3690624"],"URL":"https:\/\/doi.org\/10.1145\/3690624.3709171","relation":{},"subject":[],"published":{"date-parts":[[2025,7,20]]},"assertion":[{"value":"2025-07-20","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}