{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,6]],"date-time":"2026-07-06T17:35:12Z","timestamp":1783359312414,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":51,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001381","name":"National Research Foundation Singapore","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001381","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3758285","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:26:55Z","timestamp":1761377215000},"page":"13273-13280","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["FormFactory: An Interactive Benchmarking Suite for Multimodal Form-Filling Agents"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0513-5540","authenticated-orcid":false,"given":"Bobo","family":"Li","sequence":"first","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-5995-0077","authenticated-orcid":false,"given":"Yuheng","family":"Wang","sequence":"additional","affiliation":[{"name":"Wuhan University, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3026-6347","authenticated-orcid":false,"given":"Hao","family":"Fei","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2258-1291","authenticated-orcid":false,"given":"Juncheng","family":"Li","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8106-9768","authenticated-orcid":false,"given":"Wei","family":"Ji","sequence":"additional","affiliation":[{"name":"Nanjing University, Suzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9636-388X","authenticated-orcid":false,"given":"Mong-Li","family":"Lee","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4142-8893","authenticated-orcid":false,"given":"Wynne","family":"Hsu","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Proceedings of NeurIPS.","author":"Alayrac Jean-Baptiste","year":"2022","unstructured":"Jean-Baptiste Alayrac, Jeff Donahue, Pauline Luc, Antoine Miech, Iain Barr, Yana Hasson, Karel Lenc, Arthur Mensch, Katherine Millican, Malcolm Reynolds, Roman Ring, Eliza Rutherford, Serkan Cabi, Tengda Han, Zhitao Gong, Sina Samangooei, Marianne Monteiro, Jacob L. Menick, Sebastian Borgeaud, Andy Brock, Aida Nematzadeh, Sahand Sharifzadeh, Mikolaj Binkowski, Ricardo Barreira, Oriol Vinyals, Andrew Zisserman, and Kar\u00e9n Simonyan. 2022. Flamingo: a Visual Language Model for Few-Shot Learning. In Proceedings of NeurIPS."},{"key":"e_1_3_2_1_2_1","unstructured":"Anthropic. 2025. Claude 3.7 Sonnet System Card. (2025)."},{"key":"e_1_3_2_1_3_1","volume-title":"Qwen-VL: A frontier large vision-language model with versatile abilities. arXiv:2308.12966","author":"Bai Jinze","year":"2023","unstructured":"Jinze Bai, Shuai Bai, Shusheng Yang, Shijie Wang, Sinan Tan, Peng Wang, Junyang Lin, Chang Zhou, and Jingren Zhou. 2023. Qwen-VL: A frontier large vision-language model with versatile abilities. arXiv:2308.12966 (2023)."},{"key":"e_1_3_2_1_4_1","volume-title":"Windows Agent Arena: Evaluating Multi-Modal OS Agents at Scale. arXiv:2409.08264","author":"Bonatti Rogerio","year":"2024","unstructured":"Rogerio Bonatti, Dan Zhao, Francesco Bonacci, Dillon Dupont, Sara Abdali, Yinheng Li, Yadong Lu, Justin Wagle, Kazuhito Koishida, Arthur Bucker, Lawrence Jang, and Zack Hui. 2024. Windows Agent Arena: Evaluating Multi-Modal OS Agents at Scale. arXiv:2409.08264 (2024)."},{"key":"e_1_3_2_1_5_1","volume-title":"AMEX: Android Multi-annotation Expo Dataset for Mobile GUI Agents. arXiv:2407","author":"Chai Yuxiang","year":"2025","unstructured":"Yuxiang Chai, Siyuan Huang, Yazhe Niu, Han Xiao, Liang Liu, Dingyu Zhang, Shuai Ren, and Hongsheng Li. 2025a. AMEX: Android Multi-annotation Expo Dataset for Mobile GUI Agents. arXiv:2407.17490 (2025)."},{"key":"e_1_3_2_1_6_1","volume-title":"A3: Android Agent Arena for Mobile GUI Agents. arXiv:2501.01149","author":"Chai Yuxiang","year":"2025","unstructured":"Yuxiang Chai, Hanhao Li, Jiayu Zhang, Liang Liu, Guozhi Wang, Shuai Ren, Siyuan Huang, and Hongsheng Li. 2025b. A3: Android Agent Arena for Mobile GUI Agents. arXiv:2501.01149 (2025)."},{"key":"e_1_3_2_1_7_1","volume-title":"Proceedings of ICLR.","author":"Chen Dongping","year":"2025","unstructured":"Dongping Chen, Yue Huang, Siyuan Wu, Jingyu Tang, Huichi Zhou, Qihui Zhang, Zhigang He, Yilin Bai, Chujie Gao, Liuyi Chen, Yiqiang Li, Chenlong Wang, Yue Yu, Tianshuo Zhou, Zhen Li, Yi Gui, Yao Wan, Pan Zhou, Jianfeng Gao, and Lichao Sun. 2025a. GUI-World: A Video Benchmark and Dataset for Multimodal GUI-oriented Understanding. In Proceedings of ICLR."},{"key":"e_1_3_2_1_8_1","volume-title":"Proceedings of ICLR.","author":"Chen Jingxuan","year":"2025","unstructured":"Jingxuan Chen, Derek Yuen, Bin Xie, Yuhao Yang, Gongwei Chen, Zhihao Wu, Li Yixing, Xurui Zhou, Weiwen Liu, Shuai Wang, Kaiwen Zhou, Rui Shao, Liqiang Nie, Yasheng Wang, Jianye Hao, Jun Wang, and Kun Shao. 2025b. Spa-Bench: a comprehensive Benchmark for Smartphone Agent Evaluation. In Proceedings of ICLR."},{"key":"e_1_3_2_1_9_1","volume-title":"Shikra: Unleashing multimodal LLM's referential dialogue magic. arXiv:2306.15195","author":"Chen Ke","year":"2023","unstructured":"Ke Chen, Zhe Zhang, Wen Zeng, Richang Zhang, Feng Zhu, and Rui Zhao. 2023. Shikra: Unleashing multimodal LLM's referential dialogue magic. arXiv:2306.15195 (2023)."},{"key":"e_1_3_2_1_10_1","first-page":"9313","article-title":"SeeClick","author":"Cheng Kanzhi","year":"2024","unstructured":"Kanzhi Cheng, Qiushi Sun, Yougang Chu, Fangzhi Xu, Yantao Li, Jianbing Zhang, and Zhiyong Wu. 2024. SeeClick: Harnessing GUI Grounding for Advanced Visual GUI Agents. In Proceedings of ACL. 9313-9332.","journal-title":"Harnessing GUI Grounding for Advanced Visual GUI Agents. In Proceedings of ACL."},{"key":"e_1_3_2_1_11_1","unstructured":"Google DeepMind. 2025. Gemini 2.5: Our Most Intelligent AI Model. (2025)."},{"key":"e_1_3_2_1_12_1","volume-title":"Proceedings of NeurIPS.","author":"Deng Xiang","year":"2023","unstructured":"Xiang Deng, Yu Gu, Boyuan Zheng, Shijie Chen, Samual Stevens, Boshi Wang, Huan Sun, and Yu Su. 2023. Mind2Web: Towards a Generalist Agent for the Web. In Proceedings of NeurIPS."},{"key":"e_1_3_2_1_13_1","first-page":"14281","article-title":"CogAgent","author":"Hong Wenyi","year":"2024","unstructured":"Wenyi Hong, Weihan Wang, Qingsong Lv, Jiazheng Xu, Wenmeng Yu, Junhui Ji, Yan Wang, Zihan Wang, Yuxiao Dong, Ming Ding, and Jie Tang. 2024. CogAgent: A Visual Language Model for GUI Agents. In Proceedings of CVPR. 14281-14290.","journal-title":"A Visual Language Model for GUI Agents. In Proceedings of CVPR."},{"key":"e_1_3_2_1_14_1","volume-title":"LLaMA-Adapter: Efficient fine-tuning of language models with zero-init attention. arXiv:2303.16199","author":"Hong Weihan","year":"2023","unstructured":"Weihan Hong, Wen Wang, Jui Tseng, Mike Lewis, Xisen Shi, and Xu Chang. 2023. LLaMA-Adapter: Efficient fine-tuning of language models with zero-init attention. arXiv:2303.16199 (2023)."},{"key":"e_1_3_2_1_15_1","volume-title":"Melisa Russak, Jing Yu Koh, Kiran Kamble, Waseem AlShikh, and Ruslan Salakhutdinov.","author":"Kapoor Raghav","year":"2024","unstructured":"Raghav Kapoor, Yash Parag Butala, Melisa Russak, Jing Yu Koh, Kiran Kamble, Waseem AlShikh, and Ruslan Salakhutdinov. 2024. OmniACT: A Dataset and Benchmark for Enabling Multimodal Generalist Autonomous Agents for Desktop and Web. In Proceedings of ECCV. 161-178."},{"key":"e_1_3_2_1_16_1","volume-title":"Po-Yu Huang","author":"Koh Jing Yu","year":"2024","unstructured":"Jing Yu Koh, Robert Lo, Lawrence Jang, Vikram Duvvur, Ming Chong Lim, Po-Yu Huang, Graham Neubig, Shuyan Zhou, Russ Salakhutdinov, and Daniel Fried. 2024. VisualWebArena: Evaluating Multimodal Agents on Realistic Visual Web Tasks. In Proceedings of ACL. 881-905."},{"key":"e_1_3_2_1_17_1","first-page":"5295","article-title":"AutoWebGLM","author":"Lai Hanyu","year":"2024","unstructured":"Hanyu Lai, Xiao Liu, Iat Long Iong, Shuntian Yao, Yuxuan Chen, Pengbo Shen, Hao Yu, Hanchen Zhang, Xiaohan Zhang, Yuxiao Dong, and Jie Tang. 2024. AutoWebGLM: A Large Language Model-based Web Navigating Agent. In Proceedings of KDD. 5295-5306.","journal-title":"A Large Language Model-based Web Navigating Agent. In Proceedings of KDD."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3696410.3714920"},{"key":"e_1_3_2_1_19_1","first-page":"0","volume-title":"Proceedings of ICML","volume":"202","author":"Li Junnan","year":"1973","unstructured":"Junnan Li, Dongxu Li, Silvio Savarese, and Steven C. H. Hoi. 2023b. BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models. In Proceedings of ICML, Vol. 202. 19730-19742."},{"key":"e_1_3_2_1_20_1","first-page":"12888","article-title":"BLIP","author":"Li Junnan","year":"2022","unstructured":"Junnan Li, Dongxu Li, Caiming Xiong, and Steven C. H. Hoi. 2022. BLIP: Bootstrapping Language-Image Pre-training for Unified Vision-Language Understanding and Generation. In Proceedings of ICML. 12888-12900.","journal-title":"In Proceedings of ICML."},{"key":"e_1_3_2_1_21_1","volume-title":"VideoChat: Chat-centric video understanding. arXiv:2305.06355","author":"Li Kunyu","year":"2023","unstructured":"Kunyu Li, Yi He, Yinan Wang, Wei Li, Wen Wang, Ping Luo, Yu Wang, and Yang Qiao. 2023a. VideoChat: Chat-centric video understanding. arXiv:2305.06355 (2023)."},{"key":"e_1_3_2_1_22_1","volume-title":"Proceedings of NeurIPS.","author":"Lin Kevin Qinghong","year":"2024","unstructured":"Kevin Qinghong Lin, Linjie Li, Difei Gao, Qinchen Wu, Mingyi Yan, Zhengyuan Yang, Lijuan Wang, and Mike Zheng Shou. 2024. VideoGUI: A Benchmark for GUI Automation from Instructional Videos. In Proceedings of NeurIPS."},{"key":"e_1_3_2_1_23_1","volume-title":"Proceedings of NeurIPS.","author":"Liu Haotian","year":"2023","unstructured":"Haotian Liu, Chunyuan Li, Qingyang Wu, and Yong Jae Lee. 2023. Visual Instruction Tuning. In Proceedings of NeurIPS."},{"key":"e_1_3_2_1_24_1","unstructured":"OpenAI. 2023. GPT-4 Technical Report. arXiv:2303.08774 (2023)."},{"key":"e_1_3_2_1_25_1","volume-title":"GPT-4o System Card. arXiv:2410.21276","author":"AI.","year":"2024","unstructured":"OpenAI. 2024. GPT-4o System Card. arXiv:2410.21276 (2024)."},{"key":"e_1_3_2_1_26_1","volume-title":"WebCanvas: Benchmarking Web Agents in Online Environments. arXiv:2406.12373","author":"Pan Yichen","year":"2024","unstructured":"Yichen Pan, Dehan Kong, Sida Zhou, Cheng Cui, Yifei Leng, Bing Jiang, Hangyu Liu, Yanyi Shang, Shuyan Zhou, Tongshuang Wu, and Zhengyang Wu. 2024. WebCanvas: Benchmarking Web Agents in Online Environments. arXiv:2406.12373 (2024)."},{"key":"e_1_3_2_1_27_1","first-page":"311","article-title":"Bleu: a Method for Automatic Evaluation of Machine Translation","author":"Papineni Kishore","year":"2002","unstructured":"Kishore Papineni, Salim Roukos, Todd Ward, and Wei-Jing Zhu. 2002. Bleu: a Method for Automatic Evaluation of Machine Translation. In Proceedings of ACL. 311-318.","journal-title":"Proceedings of ACL."},{"key":"e_1_3_2_1_28_1","first-page":"8748","article-title":"Learning Transferable Visual Models From Natural Language Supervision","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021a. Learning Transferable Visual Models From Natural Language Supervision. In Proceedings of ICML. 8748-8763.","journal-title":"Proceedings of ICML."},{"key":"e_1_3_2_1_29_1","first-page":"8748","article-title":"Learning Transferable Visual Models From Natural Language Supervision","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021b. Learning Transferable Visual Models From Natural Language Supervision. In Proceedings of ICML. 8748-8763.","journal-title":"Proceedings of ICML."},{"key":"e_1_3_2_1_30_1","volume-title":"Lillicrap","author":"Rawles Christopher","year":"2023","unstructured":"Christopher Rawles, Alice Li, Daniel Rodriguez, Oriana Riva, and Timothy P. Lillicrap. 2023. AndroidInTheWild: A Large-Scale Dataset For Android Device Control. In Proceedings of NeurIPS."},{"key":"e_1_3_2_1_31_1","unstructured":"ByteDance Seed. 2024. Doubao-vision-pro-32K. (2024)."},{"key":"e_1_3_2_1_32_1","volume-title":"Proceedings of NeurIPS.","author":"Shaw Peter","year":"2023","unstructured":"Peter Shaw, Mandar Joshi, James Cohan, Jonathan Berant, Panupong Pasupat, Hexiang Hu, Urvashi Khandelwal, Kenton Lee, and Kristina Toutanova. 2023. From Pixels to UI Actions: Learning to Follow Instructions via Graphical User Interfaces. In Proceedings of NeurIPS."},{"key":"e_1_3_2_1_33_1","first-page":"3135","article-title":"World of Bits","author":"Shi Tianlin","year":"2017","unstructured":"Tianlin Shi, Andrej Karpathy, Linxi Fan, Jonathan Hernandez, and Percy Liang. 2017. World of Bits: An Open-Domain Platform for Web-Based Agents. In Proceedings of ICML. 3135-3144.","journal-title":"An Open-Domain Platform for Web-Based Agents. In Proceedings of ICML."},{"key":"e_1_3_2_1_34_1","first-page":"6699","article-title":"META-GUI","author":"Sun Liangtai","year":"2022","unstructured":"Liangtai Sun, Xingyu Chen, Lu Chen, Tianle Dai, Zichen Zhu, and Kai Yu. 2022. META-GUI: Towards Multi-modal Conversational Agents on Mobile GUI. In Proceedings of EMNLP. 6699-6712.","journal-title":"Towards Multi-modal Conversational Agents on Mobile GUI. In Proceedings of EMNLP."},{"key":"e_1_3_2_1_35_1","unstructured":"Al Sweigart. 2023. PyAutoGUI: Cross-platform GUI automation for human beings. https:\/\/pyautogui.readthedocs.io\/en\/latest\/."},{"key":"e_1_3_2_1_36_1","unstructured":"Qwen Team. 2024. Introducing Qwen-VL. (2024)."},{"key":"e_1_3_2_1_37_1","volume-title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution. CoRR","author":"Wang Peng","year":"2024","unstructured":"Peng Wang, Shuai Bai, Sinan Tan, Shijie Wang, Zhihao Fan, Jinze Bai, Keqin Chen, Xuejing Liu, Jialin Wang, Wenbin Ge, Yang Fan, Kai Dang, Mengfei Du, Xuancheng Ren, Rui Men, Dayiheng Liu, Chang Zhou, Jingren Zhou, and Junyang Lin. 2024. Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution. CoRR, Vol. abs\/2409.12191 (2024)."},{"key":"e_1_3_2_1_38_1","volume-title":"Proceedings of ICML.","author":"Wu Shengqiong","year":"2024","unstructured":"Shengqiong Wu, Hao Fei, Leigang Qu, Wei Ji, and Tat-Seng Chua. 2024. NExT-GPT: Any-to-Any Multimodal LLM. In Proceedings of ICML."},{"key":"e_1_3_2_1_39_1","unstructured":"xAITeam. 2024. Grok-3. (2024)."},{"key":"e_1_3_2_1_40_1","volume-title":"Proceedings of NeurIPS.","author":"Xie Tianbao","year":"2024","unstructured":"Tianbao Xie, Danyang Zhang, Jixuan Chen, Xiaochuan Li, Siheng Zhao, Ruisheng Cao, Toh Jing Hua, Zhoujun Cheng, Dongchan Shin, Fangyu Lei, Yitao Liu, Yiheng Xu, Shuyan Zhou, Silvio Savarese, Caiming Xiong, Victor Zhong, and Tao Yu. 2024. OSWorld: Benchmarking Multimodal Agents for Open-Ended Tasks in Real Computer Environments. In Proceedings of NeurIPS."},{"key":"e_1_3_2_1_41_1","volume-title":"MM-REACT: Prompting ChatGPT for multimodal reasoning and action. arXiv:2303.11381","author":"Yang Zhengyuan","year":"2023","unstructured":"Zhengyuan Yang, Lei Li, Jianfeng Wang, Kevin Lin, Ehsan Azarnasab, Farhad Ahmed, Zicheng Liu, Cong Liu, Michael Zeng, and Lijuan Wang. 2023. MM-REACT: Prompting ChatGPT for multimodal reasoning and action. arXiv:2303.11381 (2023)."},{"key":"e_1_3_2_1_42_1","volume-title":"Proceedings of NeurIPS.","author":"Yao Shunyu","year":"2022","unstructured":"Shunyu Yao, Howard Chen, John Yang, and Karthik Narasimhan. 2022. WebShop: Towards Scalable Real-World Web Interaction with Grounded Language Agents. In Proceedings of NeurIPS."},{"key":"e_1_3_2_1_43_1","first-page":"543","article-title":"Video-LLaMA","author":"Zhang Hang","year":"2023","unstructured":"Hang Zhang, Xin Li, and Lidong Bing. 2023. Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding. In Proceedings of EMNLP. 543-553.","journal-title":"An Instruction-tuned Audio-Visual Language Model for Video Understanding. In Proceedings of EMNLP."},{"key":"e_1_3_2_1_44_1","first-page":"12016","article-title":"Android in the Zoo","author":"Zhang Jiwen","year":"2024","unstructured":"Jiwen Zhang, Jihao Wu, Yihua Teng, Minghui Liao, Nuo Xu, Xiao Xiao, Zhongyu Wei, and Duyu Tang. 2024. Android in the Zoo: Chain-of-Action-Thought for GUI Agents. In Findings of EMNLP. 12016-12031.","journal-title":"Chain-of-Action-Thought for GUI Agents. In Findings of EMNLP."},{"key":"e_1_3_2_1_45_1","first-page":"3132","article-title":"You Only Look at Screens","author":"Zhang Zhuosheng","year":"2024","unstructured":"Zhuosheng Zhang and Aston Zhang. 2024. You Only Look at Screens: Multimodal Chain-of-Action Agents. In Findings of ACL. 3132-3149.","journal-title":"Multimodal Chain-of-Action Agents. In Findings of ACL."},{"key":"e_1_3_2_1_46_1","volume-title":"GUI Testing Arena: A Unified Benchmark for Advancing Autonomous GUI Testing Agent. arXiv:2412.18426","author":"Zhao Kangjia","year":"2024","unstructured":"Kangjia Zhao, Jiahui Song, Leigang Sha, Haozhan Shen, Zhi Chen, Tiancheng Zhao, Xiubo Liang, and Jianwei Yin. 2024b. GUI Testing Arena: A Unified Benchmark for Advancing Autonomous GUI Testing Agent. arXiv:2412.18426 (2024)."},{"key":"e_1_3_2_1_47_1","volume-title":"Proceedings of NeurIPS.","author":"Zhao Weichao","year":"2024","unstructured":"Weichao Zhao, Hao Feng, Qi Liu, Jingqun Tang, Binghong Wu, Lei Liao, Shu Wei, Yongjie Ye, Hao Liu, Wengang Zhou, Houqiang Li, and Can Huang. 2024a. TabPedia: Towards Comprehensive Visual Table Understanding with Concept Synergy. In Proceedings of NeurIPS."},{"key":"e_1_3_2_1_48_1","volume-title":"Proceedings of ICLR.","author":"Zheng Longtao","year":"2024","unstructured":"Longtao Zheng, Rundong Wang, Xinrun Wang, and Bo An. 2024b. Synapse: Trajectory-as-Exemplar Prompting with Memory for Computer Control. In Proceedings of ICLR."},{"key":"e_1_3_2_1_49_1","first-page":"9102","article-title":"Multimodal Table Understanding","author":"Zheng Mingyu","year":"2024","unstructured":"Mingyu Zheng, Xinwei Feng, Qingyi Si, Qiaoqiao She, Zheng Lin, Wenbin Jiang, and Weiping Wang. 2024a. Multimodal Table Understanding. In Proceedings of ACL. 9102-9124.","journal-title":"Proceedings of ACL."},{"key":"e_1_3_2_1_50_1","volume-title":"Proceedings of ICLR.","author":"Zhou Shuyan","year":"2024","unstructured":"Shuyan Zhou, Frank F. Xu, Hao Zhu, Xuhui Zhou, Robert Lo, Abishek Sridhar, Xianyi Cheng, Tianyue Ou, Yonatan Bisk, Daniel Fried, Uri Alon, and Graham Neubig. 2024. WebArena: A Realistic Web Environment for Building Autonomous Agents. In Proceedings of ICLR."},{"key":"e_1_3_2_1_51_1","unstructured":"Deyao Zhu Jun Chen Xiaoqian Shen Xiang Li and Mohamed Elhoseiny. 2024. MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models. In ICML."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3758285","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T05:09:24Z","timestamp":1765343364000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3758285"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":51,"alternative-id":["10.1145\/3746027.3758285","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3758285","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}