{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T06:18:25Z","timestamp":1778048305601,"version":"3.51.4"},"reference-count":40,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T00:00:00Z","timestamp":1772755200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T00:00:00Z","timestamp":1772755200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,3,6]]},"DOI":"10.1109\/wacv61042.2026.00605","type":"proceedings-article","created":{"date-parts":[[2026,5,5]],"date-time":"2026-05-05T19:59:32Z","timestamp":1778011172000},"page":"6256-6265","source":"Crossref","is-referenced-by-count":0,"title":["See, Record, Do: Automated Generation of UI Workflows from Tutorial Videos"],"prefix":"10.1109","author":[{"given":"Adam","family":"Beauchaine","sequence":"first","affiliation":[{"name":"Worcester Polytechnic Institute,Worcester,MA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Craig","family":"Shue","sequence":"additional","affiliation":[{"name":"Worcester Polytechnic Institute,Worcester,MA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","year":"2025","journal-title":"Public data set - due to anonymity requirements, this data set will be included only in a camera-ready submission"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20074-8_18"},{"key":"ref3","volume-title":"PySceneDetect: Python library and cli for video shot\/scene detection","author":"Castellano","year":"2025"},{"key":"ref4","first-page":"arXiv","article-title":"Guiworld: A dataset for gui-oriented multimodal llm-based agents","author":"Chen","year":"2024"},{"key":"ref5","article-title":"Licenses list","year":"2025"},{"key":"ref6","volume-title":"Ui automation overview - .net framework","author":"De George","year":"2021"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1145\/3126594.3126651"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.52202\/075280-1220"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ACSOS49614.2020.00038"},{"key":"ref11","article-title":"YouTube Data API v3","year":"2025"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1016\/j.jbi.2020.103571"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1145\/1982595.1982612"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1016\/j.softx.2021.100964"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1145\/3653682"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.729"},{"key":"ref18","article-title":"Reinforcement learning on web interfaces using workflow-guided exploration","author":"Liu","year":"2018"},{"key":"ref19","doi-asserted-by":"crossref","DOI":"10.20944\/preprints202501.0413.v1","article-title":"Llm-powered gui agents in phone automation: Surveying progress and prospects","author":"Liu","year":"2025"},{"key":"ref20","article-title":"Machine learning for synthetic data generation: a review","author":"Lu","year":"2023"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.741"},{"key":"ref22","article-title":"Ui-vision: A desktop-centric gui benchmark for visual perception and interaction","author":"Nayak","year":"2025"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-acl.1158"},{"key":"ref24","article-title":"NVIDIA System Management Interface (nvidia-smi)","year":"2024"},{"key":"ref25","article-title":"Ollama: Run large language models locally","author":"Developers","year":"2024"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72691-0_26"},{"key":"ref27","article-title":"Robust speech recognition via large-scale weak supervision","volume-title":"Proceedings of Machine Learning Research (PMLR)","author":"Radford"},{"key":"ref28","article-title":"Android in the wild: A large-scale dataset for android device control","author":"Rawles","year":"2023"},{"key":"ref29","article-title":"An hci-centric survey and taxonomy of human-generative-ai interactions","author":"Shi","year":"2023"},{"key":"ref30","first-page":"3135","article-title":"World of bits: An open-domain platform for web-based agents","volume-title":"International Conference on Machine Learning","author":"Shi"},{"key":"ref31","first-page":"3135","article-title":"World of bits: An open-domain platform for web-based agents","volume-title":"Proceedings of the 34th International Conference on Machine Learning, volume 70 of Proceedings of Machine Learning Research","author":"Shi"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1108\/ITP-07-2022-0519"},{"key":"ref33","article-title":"Gemma 3: A multimodal, multilingual, long-context open model","volume-title":"Technical Report arXiv:2503.19786v1, Google DeepMind","author":"Kamath","year":"2025"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1145\/1138929.1138932"},{"key":"ref35","article-title":"Os-atlas: A foundation action model for generalist gui agents","author":"Wu","year":"2024"},{"key":"ref36","article-title":"Agenttrek: Agent trajectory synthesis via guiding replay with web tutorials","author":"Xu","year":"2024"},{"key":"ref37","volume-title":"yt-dlp: A feature-rich command-line au-dio\/video downloader","year":"2020"},{"key":"ref38","article-title":"Large language model-brained gui agents: A survey","author":"Zhang","year":"2024"},{"key":"ref39","article-title":"Ufo2: The desktop agentos","author":"Zhang","year":"2025"},{"key":"ref40","article-title":"Gpt-4v (ision) is a generalist web agent, if grounded","author":"Zheng","year":"2024"}],"event":{"name":"2026 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)","location":"Tucson, AZ, USA","start":{"date-parts":[[2026,3,6]]},"end":{"date-parts":[[2026,3,10]]}},"container-title":["2026 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11491838\/11491925\/11492073.pdf?arnumber=11492073","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T05:53:59Z","timestamp":1778046839000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11492073\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,6]]},"references-count":40,"URL":"https:\/\/doi.org\/10.1109\/wacv61042.2026.00605","relation":{},"subject":[],"published":{"date-parts":[[2026,3,6]]}}}