{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T15:05:55Z","timestamp":1784300755579,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":58,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T00:00:00Z","timestamp":1783209600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,5]]},"DOI":"10.1145\/3803437.3805216","type":"proceedings-article","created":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T14:27:39Z","timestamp":1784298459000},"page":"427-438","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["From Task to Tutorial: An Automated GUI Framework for Excel Tutorial Document and Video Creation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9322-4414","authenticated-orcid":false,"given":"Yuhang","family":"Xie","sequence":"first","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-5678-1379","authenticated-orcid":false,"given":"Jian","family":"Mu","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6757-3055","authenticated-orcid":false,"given":"Xiaojun","family":"Ma","sequence":"additional","affiliation":[{"name":"Microsoft, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1304-6839","authenticated-orcid":false,"given":"Chaoyun","family":"Zhang","sequence":"additional","affiliation":[{"name":"Microsoft, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7305-1496","authenticated-orcid":false,"given":"Lu","family":"Wang","sequence":"additional","affiliation":[{"name":"Microsoft, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0322-7513","authenticated-orcid":false,"given":"Mengyu","family":"Zhou","sequence":"additional","affiliation":[{"name":"Microsoft, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-7625-8721","authenticated-orcid":false,"given":"Mugeng","family":"Liu","sequence":"additional","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8698-1860","authenticated-orcid":false,"given":"Si","family":"Qin","sequence":"additional","affiliation":[{"name":"Microsoft, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2559-2383","authenticated-orcid":false,"given":"Qingwei","family":"Lin","sequence":"additional","affiliation":[{"name":"Microsoft, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-0204-7187","authenticated-orcid":false,"given":"Saravan","family":"Rajmohan","sequence":"additional","affiliation":[{"name":"Microsoft, Redmond, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0360-6089","authenticated-orcid":false,"given":"Shi","family":"Han","sequence":"additional","affiliation":[{"name":"Microsoft, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9230-2799","authenticated-orcid":false,"given":"Dongmei","family":"Zhang","sequence":"additional","affiliation":[{"name":"Microsoft, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,17]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Agent s: An open agentic framework that uses computers like a human. arXiv preprint arXiv:2410.08164","author":"Agashe Saaket","year":"2024","unstructured":"Saaket Agashe, Jiuzhou Han, Shuyu Gan, Jiachen Yang, Ang Li, and Xin Eric Wang. 2024. Agent s: An open agentic framework that uses computers like a human. arXiv preprint arXiv:2410.08164 (2024)."},{"key":"e_1_3_2_1_2_1","unstructured":"Anthropic. 2024. Introducing Computer Use a New Claude 3.5 Sonnet and Claude 3.5 Haiku. https:\/\/www.anthropic.com\/news\/3-5-models-and-computer-use Accessed: 2024-10-26."},{"key":"e_1_3_2_1_3_1","unstructured":"Rogerio Bonatti Dan Zhao Francesco Bonacci Dillon Dupont Sara Abdali Yinheng Li Yadong Lu Justin Wagle Kazuhito Koishida Arthur Bucker et al. 2024. Windows agent arena: Evaluating multi-modal os agents at scale. arXiv preprint arXiv:2409.08264 (2024)."},{"key":"e_1_3_2_1_4_1","volume-title":"Effective educational videos: Principles and guidelines for maximizing student learning from video content. CBE\u2014Life Sciences Education","author":"Brame Cynthia J","year":"2017","unstructured":"Cynthia J Brame. 2017. Effective educational videos: Principles and guidelines for maximizing student learning from video content. CBE\u2014Life Sciences Education (2017)."},{"key":"e_1_3_2_1_5_1","volume-title":"The Nurnberg funnel: Designing minimalist instruction for practical computer skill","author":"Carroll John M","unstructured":"John M Carroll. 1990. The Nurnberg funnel: Designing minimalist instruction for practical computer skill. MIT press."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1016\/S0378-7206(96)00008-0"},{"key":"e_1_3_2_1_7_1","unstructured":"Xuetian Chen Yinghao Chen Xinfeng Yuan Zhuo Peng Lu Chen Yuekeng Li Zhoujia Zhang Yingqian Huang Leyan Huang Jiaqing Liang et al. 2025. OS-MAP: How Far Can Computer-Using Agents Go in Breadth and Depth? arXiv preprint arXiv:2507.19132 (2025)."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/3696410.3714962"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3472749.3474778"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/2380116.2380130"},{"key":"e_1_3_2_1_11_1","volume-title":"E-learning and the science of instruction","author":"Clark Ruth Colvin","unstructured":"Ruth Colvin Clark and Richard E Mayer. 2011. E-learning and the science of instruction. San Francisco: Pfeiffer."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/1964921.1964961"},{"key":"e_1_3_2_1_13_1","unstructured":"Ruomeng Ding Chaoyun Zhang Lu Wang Yong Xu Minghua Ma Wei Zhang Si Qin Saravan Rajmohan Qingwei Lin and Dongmei Zhang. [n. d.]. Everything of Thoughts: Defying the Law of Penrose Triangle for Thought Generation. ([n. d.])."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/1576246.1531372"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/1753326.1753552"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/1866029.1866054"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1057\/978-1-137-42658-1"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581641.3584069"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.3991\/ijet.v9i1.3335"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/2556288.2556986"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICSE.2007.45"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.52202\/075280-0220"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3660245"},{"key":"e_1_3_2_1_24_1","volume-title":"Infigui-r1: Advancing multimodal gui agents from reactive actors to deliberative reasoners. arXiv preprint arXiv:2504.14239","author":"Liu Yuhang","year":"2025","unstructured":"Yuhang Liu, Pengxiang Li, Congkai Xie, Xavier Hu, Xiaotian Han, Shengyu Zhang, Hongxia Yang, and Fei Wu. 2025. Infigui-r1: Advancing multimodal gui agents from reactive actors to deliberative reasoners. arXiv preprint arXiv:2504.14239 (2025)."},{"key":"e_1_3_2_1_25_1","volume-title":"Omniparser for pure vision based gui agent. arXiv preprint arXiv:2408.00203","author":"Lu Yadong","year":"2024","unstructured":"Yadong Lu, Jianwei Yang, Yelong Shen, and Ahmed Awadallah. 2024. Omniparser for pure vision based gui agent. arXiv preprint arXiv:2408.00203 (2024)."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.52202\/079017-3007"},{"key":"e_1_3_2_1_27_1","volume-title":"YouTutorial: A framework for assessing instructional online video. Technical communication quarterly 21, 1","author":"Morain Matt","year":"2012","unstructured":"Matt Morain and Jason Swarts. 2012. YouTutorial: A framework for assessing instructional online video. Technical communication quarterly 21, 1 (2012), 6\u201324."},{"key":"e_1_3_2_1_28_1","volume-title":"Computer-using Agent: Introducing a Universal Interface for AI to Interact with the Digital World. https:\/\/openai.com\/index\/computer-using-agent\/. Accessed: 2025-08-09.","author":"AI.","year":"2025","unstructured":"OpenAI. 2025. Computer-using Agent: Introducing a Universal Interface for AI to Interact with the Digital World. https:\/\/openai.com\/index\/computer-using-agent\/. Accessed: 2025-08-09."},{"key":"e_1_3_2_1_29_1","unstructured":"OpenAI. 2025. Computer-Using Agent: Introducing a universal interface for AI to interact with the digital world. (2025). https:\/\/openai.com\/index\/computer-using-agent"},{"key":"e_1_3_2_1_30_1","unstructured":"OpenAI. 2025. GPT-4.1. https:\/\/openai.com\/index\/gpt-4-1\/. Accessed: 2025-09-08."},{"key":"e_1_3_2_1_31_1","unstructured":"OpenAI. 2025. Introducing GPT-5. https:\/\/openai.com\/index\/introducing-gpt-5\/. Accessed: 2025-09-08."},{"key":"e_1_3_2_1_32_1","unstructured":"OpenAI. 2025. Introducing O3 and O4 Mini. https:\/\/openai.com\/index\/introducing-o3-and-o4-mini\/. Accessed: 2025-09-08."},{"key":"e_1_3_2_1_33_1","unstructured":"OpenAI. 2025. OpenAI API Reference. https:\/\/platform.openai.com\/docs\/api-reference\/chat. Accessed: 2025-09-08."},{"key":"e_1_3_2_1_34_1","volume-title":"Benjamin Van Durme, and Elnaz Nouri","author":"Payan Justin","year":"2023","unstructured":"Justin Payan, Swaroop Mishra, Mukul Singh, Carina Negreanu, Christian Poelitz, Chitta Baral, Subhro Roy, Rasika Chakravarthy, Benjamin Van Durme, and Elnaz Nouri. 2023. Instructexcel: A benchmark for natural language instruction in excel. arXiv preprint arXiv:2310.14495 (2023)."},{"key":"e_1_3_2_1_35_1","volume-title":"Business analytics: The art of modeling with spreadsheets","author":"Powell Stephen G","unstructured":"Stephen G Powell and Kenneth R Baker. 2019. Business analytics: The art of modeling with spreadsheets. John Wiley & Sons."},{"key":"e_1_3_2_1_36_1","volume-title":"Sriparna Saha, Vinija Jain, Samrat Mondal, and Aman Chadha.","author":"Sahoo Pranab","year":"2024","unstructured":"Pranab Sahoo, Ayush Kumar Singh, Sriparna Saha, Vinija Jain, Samrat Mondal, and Aman Chadha. 2024. A systematic survey of prompt engineering in large language models: Techniques and applications. arXiv preprint arXiv:2402.07927 (2024)."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"crossref","unstructured":"Jeff Sauro and James R Lewis. 2016. Quantifying the user experience: Practical statistics for user research. Morgan Kaufmann.","DOI":"10.1016\/B978-0-12-802308-2.00002-3"},{"key":"e_1_3_2_1_38_1","unstructured":"Simeng Sun Aston Zhang Junxian He Zhiyuan Jin Tong Zhao Bang Chen Wenyi Liu Yelong Shen Shuai Wang Haoyu Wang et al. 2023. Principled instructions are all you need for questioning LLMs. arXiv preprint arXiv:2306.01577 (2023)."},{"key":"e_1_3_2_1_39_1","first-page":"195","article-title":"New modes of help: Best practices for instructional video","volume":"59","author":"Swarts Jason","year":"2012","unstructured":"Jason Swarts. 2012. New modes of help: Best practices for instructional video. Technical Communication 59, 3 (2012), 195\u2013206.","journal-title":"Technical Communication"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-84800-031-5_21"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445721"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3313831.3376759"},{"key":"e_1_3_2_1_43_1","volume-title":"Eight guidelines for the design of instructional videos for software training. Technical communication 60, 3","author":"der Meij Hans Van","year":"2013","unstructured":"Hans Van der Meij and Jan van der Meij. 2013. Eight guidelines for the design of instructional videos for software training. Technical communication 60, 3 (2013), 205\u2013228."},{"key":"e_1_3_2_1_44_1","unstructured":"Lu Wang Fangkai Yang Chaoyun Zhang Junting Lu Jiaxu Qian Shilin He Pu Zhao Bo Qiao Ray Huang Si Qin et al. 2024. Large action models: From inception to implementation. arXiv preprint arXiv:2412.10047 (2024)."},{"key":"e_1_3_2_1_45_1","volume-title":"Jin Xu, Victor R\u00fchle, and Saravan Rajmohan.","author":"Wang Weixuan","year":"2025","unstructured":"Weixuan Wang, Dongge Han, Daniel Madrigal Diaz, Jin Xu, Victor R\u00fchle, and Saravan Rajmohan. 2025. Odysseybench: Evaluating llm agents on long-horizon complex office application workflows. arXiv preprint arXiv:2508.09124 (2025)."},{"key":"e_1_3_2_1_46_1","volume-title":"Denny Zhou, et al.","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Fei Xia, Ed Chi, Quoc V Le, Denny Zhou, et al. 2022. Chain-of-thought prompting elicits reasoning in large language models. Advances in neural information processing systems 35 (2022), 24824\u201324837."},{"key":"e_1_3_2_1_47_1","volume-title":"Proceedings of the 28th ACM Joint Meeting on European Software Engineering Conference and Symposium on the Foundations of Software Engineering. 2122\u20132133","author":"White Jules","year":"2023","unstructured":"Jules White, Vibhu Rastogi, Sam H Robinson, Jack Johnson, Neel Mitchell, and Douglas C Schmidt. 2023. A prompt pattern catalog to enhance prompt engineering with chatgpt. In Proceedings of the 28th ACM Joint Meeting on European Software Engineering Conference and Symposium on the Foundations of Software Engineering. 2122\u20132133."},{"key":"e_1_3_2_1_48_1","unstructured":"Qianhui Wu Kanzhi Cheng Rui Yang Chaoyun Zhang Jianwei Yang Huiqiang Jiang Jian Mu Baolin Peng Bo Qiao Reuben Tan et al. 2025. GUI-Actor: Coordinate-Free Visual Grounding for GUI Agents. arXiv preprint arXiv:2506.03143 (2025)."},{"key":"e_1_3_2_1_49_1","volume-title":"Set-of-mark prompting unleashes extraordinary visual grounding in gpt-4v. arXiv preprint arXiv:2310.11441","author":"Yang Jianwei","year":"2023","unstructured":"Jianwei Yang, Hao Zhang, Feng Li, Xueyan Zou, Chunyuan Li, and Jianfeng Gao. 2023. Set-of-mark prompting unleashes extraordinary visual grounding in gpt-4v. arXiv preprint arXiv:2310.11441 (2023)."},{"key":"e_1_3_2_1_50_1","volume-title":"2014 IEEE 14th international conference on advanced learning technologies. IEEE, 44\u201348","author":"Fahmy Yousef Ahmed Mohamed","year":"2014","unstructured":"Ahmed Mohamed Fahmy Yousef, Mohamed Amine Chatti, Ulrik Schroeder, and Marold Wosnitza. 2014. What drives a successful MOOC? An empirical examination of criteria to assure design quality of MOOCs. In 2014 IEEE 14th international conference on advanced learning technologies. IEEE, 44\u201348."},{"key":"e_1_3_2_1_51_1","volume-title":"GUI Agents: Divergence and Convergence. In ICML 2025 Workshop on Computer Use Agents.","author":"Zhang Chaoyun","unstructured":"Chaoyun Zhang, Shilin He, Liqun Li, Si Qin, Yu Kang, Qingwei Lin, Saravan Rajmohan, and Dongmei Zhang. [n. d.]. API Agents vs. GUI Agents: Divergence and Convergence. In ICML 2025 Workshop on Computer Use Agents."},{"key":"e_1_3_2_1_52_1","unstructured":"Chaoyun Zhang Shilin He Jiaxu Qian Bowen Li Liqun Li Si Qin Yu Kang Minghua Ma Guyue Liu Qingwei Lin et al. 2024. Large language model-brained gui agents: A survey. arXiv preprint arXiv:2411.18279 (2024)."},{"key":"e_1_3_2_1_53_1","unstructured":"Chaoyun Zhang He Huang Chiming Ni Jian Mu Si Qin Shilin He Lu Wang Fangkai Yang Pu Zhao Chao Du et al. 2025. Ufo2: The desktop agentos. arXiv preprint arXiv:2504.14603 (2025)."},{"key":"e_1_3_2_1_54_1","volume-title":"Ufo: A ui-focused agent for windows os interaction. arXiv preprint arXiv:2402.07939","author":"Zhang Chaoyun","year":"2024","unstructured":"Chaoyun Zhang, Liqun Li, Shilin He, Xu Zhang, Bo Qiao, Si Qin, Minghua Ma, Yu Kang, Qingwei Lin, Saravan Rajmohan, et al. 2024. Ufo: A ui-focused agent for windows os interaction. arXiv preprint arXiv:2402.07939 (2024)."},{"key":"e_1_3_2_1_55_1","volume-title":"Vem: Environment-free exploration for training gui agent with value environment model. arXiv preprint arXiv:2502.18906","author":"Zheng Jiani","year":"2025","unstructured":"Jiani Zheng, Lu Wang, Fangkai Yang, Chaoyun Zhang, Lingrui Mei, Wenjie Yin, Qingwei Lin, Dongmei Zhang, Saravan Rajmohan, and Qi Zhang. 2025. Vem: Environment-free exploration for training gui agent with value environment model. arXiv preprint arXiv:2502.18906 (2025)."},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1145\/3472749.3474812"},{"key":"e_1_3_2_1_57_1","volume-title":"Proceedings of 19th European Conference on Computer-Supported Cooperative Work. European Society for Socially Embedded Technologies (EUSSET).","author":"Zhu Qingxiaoyang","year":"2021","unstructured":"Qingxiaoyang Zhu and Hao-Chuan Wang. 2021. Is a GIF Worth a Thousand Words? Understanding the Use of Dynamic Graphical Illustrations for Procedural Knowledge Sharing on wikiHow. In Proceedings of 19th European Conference on Computer-Supported Cooperative Work. European Society for Socially Embedded Technologies (EUSSET)."},{"key":"e_1_3_2_1_58_1","volume-title":"SheetMind: An End-to-End LLM-Powered Multi-Agent Framework for Spreadsheet Automation. arXiv preprint arXiv:2506.12339","author":"Zhu Ruiyan","year":"2025","unstructured":"Ruiyan Zhu, Xi Cheng, Ke Liu, Brian Zhu, Daniel Jin, Neeraj Parihar, Zhoutian Xu, and Oliver Gao. 2025. SheetMind: An End-to-End LLM-Powered Multi-Agent Framework for Spreadsheet Automation. arXiv preprint arXiv:2506.12339 (2025)."}],"event":{"name":"FSE Companion '26: 34th ACM International Conference on the Foundations of Software Engineering","location":"Concordia University Montreal QC Canada","acronym":"FSE Companion '26","sponsor":["SIGSOFT ACM Special Interest Group on Software Engineering"]},"container-title":["Proceedings of the 34th ACM International Conference on the Foundations of Software Engineering"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3803437.3805216","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T14:29:56Z","timestamp":1784298596000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3803437.3805216"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,5]]},"references-count":58,"alternative-id":["10.1145\/3803437.3805216","10.1145\/3803437"],"URL":"https:\/\/doi.org\/10.1145\/3803437.3805216","relation":{},"subject":[],"published":{"date-parts":[[2026,7,5]]},"assertion":[{"value":"2026-07-17","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}