{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T20:00:37Z","timestamp":1765310437751,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":52,"publisher":"ACM","funder":[{"name":"National Research Foundation, Singapore under its AI Singapore Programme","award":["AISG3-RP-2022-030"],"award-info":[{"award-number":["AISG3-RP-2022-030"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755150","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:30:51Z","timestamp":1761377451000},"page":"3683-3692","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["GUI-Narrator: Detecting and Captioning Computer GUI Actions"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-1722-2579","authenticated-orcid":false,"given":"Qinchen","family":"Wu","sequence":"first","affiliation":[{"name":"Show Lab, National University of Singapore, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8494-3492","authenticated-orcid":false,"given":"Difei","family":"Gao","sequence":"additional","affiliation":[{"name":"Show Lab, National University of Singapore, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-6779-3435","authenticated-orcid":false,"given":"Qinghong","family":"Lin","sequence":"additional","affiliation":[{"name":"Show Lab, National University of Singapore, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9075-7626","authenticated-orcid":false,"given":"Zhuoyu","family":"Wu","sequence":"additional","affiliation":[{"name":"Show Lab, National University of Singapore, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7681-2166","authenticated-orcid":false,"given":"Mike Zheng","family":"Shou","sequence":"additional","affiliation":[{"name":"Show Lab, National University of Singapore, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Anthropic. 2024. Claude 3.5 Sonnet. https:\/\/www.anthropic.com\/news\/claude-3-5-sonnet. Accessed: 2025-02-02."},{"key":"e_1_3_2_1_2_1","unstructured":"Jinze Bai Shuai Bai Yunfei Chu Zeyu Cui Kai Dang Xiaodong Deng Yang Fan Wenbin Ge Yu Han Fei Huang et al. 2023. Qwen technical report. arXiv preprint arXiv:2309.16609 (2023)."},{"key":"e_1_3_2_1_3_1","volume-title":"GUIDE: Graphical User Interface Data for Execution. arXiv preprint arXiv:2404.16048","author":"Chawla Rajat","year":"2024","unstructured":"Rajat Chawla, Adarsh Jha, Muskaan Kumar, Mukunda NS, and Ishaan Bhola. 2024. GUIDE: Graphical User Interface Data for Execution. arXiv preprint arXiv:2404.16048 (2024)."},{"key":"e_1_3_2_1_4_1","unstructured":"Dongping Chen Yue Huang Siyuan Wu Jingyu Tang Liuyi Chen Yilin Bai Zhigang He Chenlong Wang Huichi Zhou Yiqiang Li et al. 2024b. GUI-WORLD: A Dataset for GUI-oriented Multimodal LLM-based Agents. arXiv preprint arXiv:2406.10819 (2024)."},{"key":"e_1_3_2_1_5_1","unstructured":"Wentong Chen Junbo Cui Jinyi Hu Yujia Qin Junjie Fang Yue Zhao Chongyi Wang Jun Liu Guirong Chen Yupeng Huo et al. 2024a. GUICourse: From General Vision Language Models to Versatile GUI Agents. arXiv preprint arXiv:2406.11317 (2024)."},{"key":"e_1_3_2_1_6_1","unstructured":"Zhe Chen Weiyun Wang Yue Cao Yangzhou Liu Zhangwei Gao Erfei Cui Jinguo Zhu Shenglong Ye Hao Tian Zhaoyang Liu et al. 2024c. Expanding Performance Boundaries of Open-Source Multimodal Models with Model Data and Test-Time Scaling. arXiv preprint arXiv:2412.05271 (2024)."},{"key":"e_1_3_2_1_7_1","volume-title":"HALC: Object Hallucination Reduction via Adaptive Focal-Contrast Decoding. arXiv:2403.00425 [cs.CV] https:\/\/arxiv.org\/abs\/2403.00425","author":"Chen Zhaorun","year":"2024","unstructured":"Zhaorun Chen, Zhuokai Zhao, Hongyin Luo, Huaxiu Yao, Bo Li, and Jiawei Zhou. 2024d. HALC: Object Hallucination Reduction via Adaptive Focal-Contrast Decoding. arXiv:2403.00425 [cs.CV] https:\/\/arxiv.org\/abs\/2403.00425"},{"key":"e_1_3_2_1_8_1","volume-title":"Seeclick: Harnessing gui grounding for advanced visual gui agents. arXiv preprint arXiv:2401.10935","author":"Cheng Kanzhi","year":"2024","unstructured":"Kanzhi Cheng, Qiushi Sun, Yougang Chu, Fangzhi Xu, Yantao Li, Jianbing Zhang, and Zhiyong Wu. 2024. Seeclick: Harnessing gui grounding for advanced visual gui agents. arXiv preprint arXiv:2401.10935 (2024)."},{"key":"e_1_3_2_1_9_1","first-page":"28091","volume-title":"Levine (Eds.)","volume":"36","author":"Deng Xiang","year":"2023","unstructured":"Xiang Deng, Yu Gu, Boyuan Zheng, Shijie Chen, Sam Stevens, Boshi Wang, Huan Sun, and Yu Su. 2023. Mind2Web: Towards a Generalist Agent for the Web, A. Oh, T. Naumann, A. Globerson, K. Saenko, M. Hardt, and S. Levine (Eds.), Vol. 36. Curran Associates, Inc., 28091-28114. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2023\/file\/5950bf290a1570ea401bf98882128160-Paper-Datasets_and_Benchmarks.pdf"},{"key":"e_1_3_2_1_10_1","volume-title":"Jepson","author":"Dvornik Nikita","year":"2023","unstructured":"Nikita Dvornik, Isma Hadji, Ran Zhang, Konstantinos G. Derpanis, Animesh Garg, Richard P. Wildes, and Allan D. Jepson. 2023. StepFormer: Self-supervised Step Discovery and Localization in Instructional Videos. arXiv:2304.13265 [cs.CV] https:\/\/arxiv.org\/abs\/2304.13265"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"crossref","unstructured":"Difei Gao Lei Ji Zechen Bai Mingyu Ouyang Peiran Li Dongxing Mao Qinchen Wu Weichen Zhang Peiyi Wang Xiangwu Guo et al. 2024. AssistGUI: Task-Oriented PC Graphical User Interface Automation. (2024) 13289-13298.","DOI":"10.1109\/CVPR52733.2024.01262"},{"key":"e_1_3_2_1_12_1","volume-title":"Joya Chen, Zihan Fan, and Mike Zheng Shou.","author":"Gao Difei","year":"2023","unstructured":"Difei Gao, Lei Ji, Luowei Zhou, Kevin Qinghong Lin, Joya Chen, Zihan Fan, and Mike Zheng Shou. 2023. Assistgpt: A general multi-modal assistant that can plan, execute, inspect, and learn. arXiv preprint arXiv:2306.08640 (2023)."},{"key":"e_1_3_2_1_13_1","volume-title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents. arXiv preprint arXiv:2410.05243","author":"Gou Boyu","year":"2024","unstructured":"Boyu Gou, Ruohan Wang, Boyuan Zheng, Yanan Xie, Cheng Chang, Yiheng Shu, Huan Sun, and Yu Su. 2024. Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents. arXiv preprint arXiv:2410.05243 (2024). https:\/\/arxiv.org\/abs\/2410.05243"},{"key":"e_1_3_2_1_14_1","volume-title":"Cogagent: A visual language model for gui agents.","author":"Hong Wenyi","year":"2024","unstructured":"Wenyi Hong, Weihan Wang, Qingsong Lv, Jiazheng Xu, Wenmeng Yu, Junhui Ji, Yan Wang, Zihan Wang, Yuxiao Dong, Ming Ding, et al., 2024. Cogagent: A visual language model for gui agents. (2024), 14281-14290."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73039-9_12"},{"key":"e_1_3_2_1_16_1","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Kim Geunwoo","year":"2024","unstructured":"Geunwoo Kim, Pierre Baldi, and Stephen McAleer. 2024. Language models can solve computer tasks. Advances in Neural Information Processing Systems, Vol. 36 (2024)."},{"key":"e_1_3_2_1_17_1","volume-title":"Po-Yu Huang","author":"Koh Jing Yu","year":"2024","unstructured":"Jing Yu Koh, Robert Lo, Lawrence Jang, Vikram Duvvur, Ming Chong Lim, Po-Yu Huang, Graham Neubig, Shuyan Zhou, Ruslan Salakhutdinov, and Daniel Fried. 2024. VisualWebArena: EVALUATING MULTIMODAL. arXiv preprint arXiv:2401.13649 (2024)."},{"key":"e_1_3_2_1_18_1","volume-title":"Swinbert: End-to-end transformers with sparse attention for video captioning.","author":"Lin Kevin","year":"2022","unstructured":"Kevin Lin, Linjie Li, Chung-Ching Lin, Faisal Ahmed, Zhe Gan, Zicheng Liu, Yumao Lu, and Lijuan Wang. 2022. Swinbert: End-to-end transformers with sparse attention for video captioning. (2022), 17949-17958."},{"key":"e_1_3_2_1_19_1","volume-title":"ShowUI: One Vision-Language-Action Model for GUI Visual Agent. arXiv preprint arXiv:2411.17465","author":"Lin Kevin Qinghong","year":"2024","unstructured":"Kevin Qinghong Lin, Linjie Li, Difei Gao, Zhengyuan Yang, Shiwei Wu, Zechen Bai, Weixian Lei, Lijuan Wang, and Mike Zheng Shou. 2024. ShowUI: One Vision-Language-Action Model for GUI Visual Agent. arXiv preprint arXiv:2411.17465 (2024)."},{"key":"e_1_3_2_1_20_1","volume-title":"Visual instruction tuning. Advances in neural information processing systems","author":"Liu Haotian","year":"2024","unstructured":"Haotian Liu, Chunyuan Li, Qingyang Wu, and Yong Jae Lee. 2024b. Visual instruction tuning. Advances in neural information processing systems, Vol. 36 (2024)."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"crossref","unstructured":"Zhe Liu Chunyang Chen Junjie Wang Xing Che Yuekai Huang Jun Hu and Qing Wang. 2023a. Fill in the blank: Context-aware automated text input generation for mobile gui testing. (2023) 1355-1367.","DOI":"10.1109\/ICSE48619.2023.00119"},{"key":"e_1_3_2_1_22_1","volume-title":"Chatting with gpt-3 for zero-shot human-like mobile automated gui testing. arXiv preprint arXiv:2305.09434","author":"Liu Zhe","year":"2023","unstructured":"Zhe Liu, Chunyang Chen, Junjie Wang, Mengzhuo Chen, Boyu Wu, Xing Che, Dandan Wang, and Qing Wang. 2023b. Chatting with gpt-3 for zero-shot human-like mobile automated gui testing. arXiv preprint arXiv:2305.09434 (2023)."},{"key":"e_1_3_2_1_23_1","unstructured":"Zuyan Liu Yuhao Dong Yongming Rao Jie Zhou and Jiwen Lu. 2024a. Chain-of-Spot: Interactive Reasoning Improves Large Vision-Language Models. arXiv:2403.12966 [cs.CV] https:\/\/arxiv.org\/abs\/2403.12966"},{"key":"e_1_3_2_1_24_1","volume-title":"Comprehensive CognitiAGENTS ON REALISTIC VISUAL WEB TASKSve LLM Agent for Smartphone GUI Automation. arXiv preprint arXiv:2402.11941","author":"Ma Xinbei","year":"2024","unstructured":"Xinbei Ma, Zhuosheng Zhang, and Hai Zhao. 2024. Comprehensive CognitiAGENTS ON REALISTIC VISUAL WEB TASKSve LLM Agent for Smartphone GUI Automation. arXiv preprint arXiv:2402.11941 (2024)."},{"key":"e_1_3_2_1_25_1","volume-title":"Improving web element localization by using a large language model. arXiv preprint arXiv:2310.02046","author":"Nass Michel","year":"2023","unstructured":"Michel Nass, Emil Alegroth, and Robert Feldt. 2023. Improving web element localization by using a large language model. arXiv preprint arXiv:2310.02046 (2023)."},{"key":"e_1_3_2_1_26_1","volume-title":"ScreenAgent: A Vision Language Model-driven Computer Control Agent. arXiv preprint arXiv:2402.07945","author":"Niu Runliang","year":"2024","unstructured":"Runliang Niu, Jindong Li, Shiqi Wang, Yali Fu, Xiyu Hu, Xueyuan Leng, He Kong, Yi Chang, and Qi Wang. 2024. ScreenAgent: A Vision Language Model-driven Computer Control Agent. arXiv preprint arXiv:2402.07945 (2024)."},{"key":"e_1_3_2_1_27_1","unstructured":"OpenAI. 2024. Hello GPT-4o. https:\/\/openai.com\/index\/hello-gpt-4o\/ Accessed: 2024-05-21."},{"key":"e_1_3_2_1_28_1","unstructured":"Boxiao Pan Haoye Cai De-An Huang Kuan-Hui Lee Adrien Gaidon Ehsan Adeli and Juan Carlos Niebles. 2020. Spatio-temporal graph for video captioning with knowledge distillation. (2020) 10870-10879."},{"key":"e_1_3_2_1_29_1","volume-title":"AutoTask: Executing Arbitrary Voice Commands by Exploring and Learning from Mobile GUI. arXiv preprint arXiv:2312.16062","author":"Pan Lihang","year":"2023","unstructured":"Lihang Pan, Bowen Wang, Chun Yu, Yuxuan Chen, Xiangyu Zhang, and Yuanchun Shi. 2023. AutoTask: Executing Arbitrary Voice Commands by Exploring and Learning from Mobile GUI. arXiv preprint arXiv:2312.16062 (2023)."},{"key":"e_1_3_2_1_30_1","unstructured":"Machel Reid Nikolay Savinov Denis Teplyashin Dmitry Lepikhin Timothy Lillicrap Jean-baptiste Alayrac Radu Soricut Angeliki Lazaridou Orhan Firat Julian Schrittwieser et al. 2024. Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context. arXiv preprint arXiv:2403.05530 (2024)."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-11752-2_15"},{"key":"e_1_3_2_1_32_1","volume-title":"Charades-ego: A large-scale dataset of paired third and first person videos. arXiv preprint arXiv:1804.09626","author":"Sigurdsson Gunnar A","year":"2018","unstructured":"Gunnar A Sigurdsson, Abhinav Gupta, Cordelia Schmid, Ali Farhadi, and Karteek Alahari. 2018. Charades-ego: A large-scale dataset of paired third and first person videos. arXiv preprint arXiv:1804.09626 (2018)."},{"key":"e_1_3_2_1_33_1","volume-title":"META-GUI: towards multi-modal conversational agents on mobile GUI. arXiv preprint arXiv:2205.11029","author":"Sun Liangtai","year":"2022","unstructured":"Liangtai Sun, Xingyu Chen, Lu Chen, Tianle Dai, Zichen Zhu, and Kai Yu. 2022. META-GUI: towards multi-modal conversational agents on mobile GUI. arXiv preprint arXiv:2205.11029 (2022)."},{"key":"e_1_3_2_1_34_1","unstructured":"Ultralytics. 2024. YOLOv8 Documentation. https:\/\/docs.ultralytics.com\/models\/yolov8\/. Accessed: 2025-02-12."},{"key":"e_1_3_2_1_35_1","volume-title":"UGIF: UI Grounded Instruction Following.","author":"Venkatesh Sagar Gubbi","year":"2023","unstructured":"Sagar Gubbi Venkatesh, Partha Talukdar, and Srini Narayanan. 2023. UGIF: UI Grounded Instruction Following. (2023). arXiv:2211.07615 [cs.CL]"},{"key":"e_1_3_2_1_36_1","unstructured":"Jianqiang Wan Sibo Song Wenwen Yu Yuliang Liu Wenqing Cheng Fei Huang Xiang Bai Cong Yao and Zhibo Yang. 2024. OmniParser: A Unified Framework for Text Spotting Key Information Extraction and Table Recognition. arXiv:2403.19128 [cs.CV] https:\/\/arxiv.org\/abs\/2403.19128"},{"key":"e_1_3_2_1_37_1","volume-title":"Grounded-videollm: Sharpening fine-grained temporal grounding in video large language models. arXiv preprint arXiv:2410.03290","author":"Wang Haibo","year":"2024","unstructured":"Haibo Wang, Zhiyang Xu, Yu Cheng, Shizhe Diao, Yufan Zhou, Yixin Cao, Qifan Wang, Weifeng Ge, and Lifu Huang. 2024c. Grounded-videollm: Sharpening fine-grained temporal grounding in video large language models. arXiv preprint arXiv:2410.03290 (2024)."},{"key":"e_1_3_2_1_38_1","volume-title":"Understanding User Experience in Large Language Model Interactions. arXiv preprint arXiv:2401.08329","author":"Wang Jiayin","year":"2024","unstructured":"Jiayin Wang, Weizhi Ma, Peijie Sun, Min Zhang, and Jian-Yun Nie. 2024b. Understanding User Experience in Large Language Model Interactions. arXiv preprint arXiv:2401.08329 (2024)."},{"key":"e_1_3_2_1_39_1","volume-title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution. arXiv preprint arXiv:2409.12191","author":"Wang Peng","year":"2024","unstructured":"Peng Wang, Shuai Bai, Sinan Tan, Shijie Wang, Zhihao Fan, Jinze Bai, Keqin Chen, Xuejing Liu, Jialin Wang, Wenbin Ge, Yang Fan, Kai Dang, Mengfei Du, Xuancheng Ren, Rui Men, Dayiheng Liu, Chang Zhou, Jingren Zhou, and Junyang Lin. 2024a. Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution. arXiv preprint arXiv:2409.12191 (2024)."},{"key":"e_1_3_2_1_40_1","volume-title":"Shiqi Jiang, Yunhao Liu, Yaqin Zhang, and Yunxin Liu.","author":"Wen Hao","year":"2024","unstructured":"Hao Wen, Yuanchun Li, Guohong Liu, Shanhui Zhao, Tao Yu, Toby Jia-Jun Li, Shiqi Jiang, Yunhao Liu, Yaqin Zhang, and Yunxin Liu. 2024. AutoDroid: LLM-powered Task Automation in Android. (2024)."},{"key":"e_1_3_2_1_41_1","volume-title":"Zhoujun Cheng, Dongchan Shin, Fangyu Lei, et al.","author":"Xie Tianbao","year":"2024","unstructured":"Tianbao Xie, Danyang Zhang, Jixuan Chen, Xiaochuan Li, Siheng Zhao, Ruisheng Cao, Toh Jing Hua, Zhoujun Cheng, Dongchan Shin, Fangyu Lei, et al., 2024. OSWorld: Benchmarking Multimodal Agents for Open-Ended Tasks in Real Computer Environments. arXiv preprint arXiv:2404.07972 (2024)."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123427"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.571"},{"key":"e_1_3_2_1_44_1","volume-title":"Aguvis: Unified Pure Vision Agents for Autonomous GUI Interaction.","author":"Xu Yiheng","year":"2024","unstructured":"Yiheng Xu, Zekun Wang, Junli Wang, Dunjie Lu, Tianbao Xie, Amrita Saha, Doyen Sahoo, Tao Yu, and Caiming Xiong. 2024. Aguvis: Unified Pure Vision Agents for Autonomous GUI Interaction. (2024). https:\/\/arxiv.org\/abs\/2412.04454"},{"key":"e_1_3_2_1_45_1","unstructured":"An Yan Zhengyuan Yang Wanrong Zhu Kevin Lin Linjie Li Jianfeng Wang Jianwei Yang Yiwu Zhong Julian McAuley Jianfeng Gao et al. 2023. Gpt-4v in wonderland: Large multimodal models for zero-shot smartphone gui navigation. arXiv preprint arXiv:2311.07562 (2023)."},{"key":"e_1_3_2_1_46_1","volume-title":"Antoine Miech, Jordi Pont-Tuset, Ivan Laptev, Josef Sivic, and Cordelia Schmid.","author":"Yang Antoine","year":"2023","unstructured":"Antoine Yang, Arsha Nagrani, Paul Hongsuck Seo, Antoine Miech, Jordi Pont-Tuset, Ivan Laptev, Josef Sivic, and Cordelia Schmid. 2023b. Vid2seq: Large-scale pretraining of a visual language model for dense video captioning. (2023), 10714-10726."},{"key":"e_1_3_2_1_47_1","volume-title":"Appagent: Multimodal agents as smartphone users. arXiv preprint arXiv:2312.13771","author":"Yang Zhao","year":"2023","unstructured":"Zhao Yang, Jiaxuan Liu, Yucheng Han, Xin Chen, Zebiao Huang, Bin Fu, and Gang Yu. 2023a. Appagent: Multimodal agents as smartphone users. arXiv preprint arXiv:2312.13771 (2023)."},{"key":"e_1_3_2_1_48_1","volume-title":"You only look at screens: Multimodal chain-of-action agents. arXiv preprint arXiv:2309.11436","author":"Zhan Zhuosheng","year":"2023","unstructured":"Zhuosheng Zhan and Aston Zhang. 2023. You only look at screens: Multimodal chain-of-action agents. arXiv preprint arXiv:2309.11436 (2023)."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"crossref","unstructured":"Chenlin Zhang Jianxin Wu and Yin Li. 2022. ActionFormer: Localizing Moments of Actions with Transformers. arXiv:2202.07925 [cs.CV] https:\/\/arxiv.org\/abs\/2202.07925","DOI":"10.1007\/978-3-031-19772-7_29"},{"key":"e_1_3_2_1_50_1","volume-title":"Android in the Zoo: Chain-of-Action-Thought for GUI Agents. arXiv preprint arXiv:2403.02713","author":"Zhang Jiwen","year":"2024","unstructured":"Jiwen Zhang, Jihao Wu, Yihua Teng, Minghui Liao, Nuo Xu, Xiao Xiao, Zhongyu Wei, and Duyu Tang. 2024. Android in the Zoo: Chain-of-Action-Thought for GUI Agents. arXiv preprint arXiv:2403.02713 (2024)."},{"key":"e_1_3_2_1_51_1","volume-title":"AgentStudio: A Toolkit for Building General Virtual Agents. arXiv preprint arXiv:2403.17918","author":"Zheng Longtao","year":"2024","unstructured":"Longtao Zheng, Zhiyuan Huang, Zhenghai Xue, Xinrun Wang, Bo An, and Shuicheng Yan. 2024. AgentStudio: A Toolkit for Building General Virtual Agents. arXiv preprint arXiv:2403.17918 (2024)."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.12342"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755150","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:57:08Z","timestamp":1765310228000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755150"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":52,"alternative-id":["10.1145\/3746027.3755150","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755150","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}