{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T07:58:16Z","timestamp":1776931096936,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":13,"publisher":"ACM","funder":[{"name":"The Japan Society for the Promotion of Science","award":["JP24K02942, JP23K21676, JP23K11141, and JP23K11211"],"award-info":[{"award-number":["JP24K02942, JP23K21676, JP23K11141, and JP23K11211"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,15]]},"DOI":"10.1145\/3757376.3771402","type":"proceedings-article","created":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T05:46:28Z","timestamp":1765345588000},"page":"1-4","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Lost in the Interface: How Structured UI Complexity Challenges Large Vision Language Models in Games"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-6012-9183","authenticated-orcid":false,"given":"Xiang","family":"Li","sequence":"first","affiliation":[{"name":"Hokkaido University, Sapporo, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4474-3995","authenticated-orcid":false,"given":"Ren","family":"Togo","sequence":"additional","affiliation":[{"name":"Hokkaido University, Sapporo, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8039-3462","authenticated-orcid":false,"given":"Keisuke","family":"Maeda","sequence":"additional","affiliation":[{"name":"Hokkaido University, Sapporo, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5332-8112","authenticated-orcid":false,"given":"Takahiro","family":"Ogawa","sequence":"additional","affiliation":[{"name":"Hokkaido University, Sapporo, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1496-1761","authenticated-orcid":false,"given":"Miki","family":"Haseyama","sequence":"additional","affiliation":[{"name":"Hokkaido University, Sapporo, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,12,14]]},"reference":[{"key":"e_1_3_3_2_2_1","unstructured":"Anthropic.2025. Introducing Claude 4. https:\/\/www.anthropic.com\/claude (2025)."},{"key":"e_1_3_3_2_3_1","unstructured":"Jinze Bai Shuai Bai Yunfei Chu Zeyu Cui Kai Dang Xiaodong Deng Yang Fan Wenbin Ge Yu Han Fei Huang et\u00a0al. 2023. Qwen technical report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2309.16609 (2023)."},{"key":"e_1_3_3_2_4_1","first-page":"2048","volume-title":"International conference on machine learning","author":"Cobbe Karl","year":"2020","unstructured":"Karl Cobbe, Chris Hesse, Jacob Hilton, and John Schulman. 2020. Leveraging procedural generation to benchmark reinforcement learning. In International conference on machine learning. PMLR, 2048\u20132056."},{"key":"e_1_3_3_2_5_1","doi-asserted-by":"crossref","unstructured":"Maxime Delmas Lo\u00efc Caroux and C\u00e9line Lemercier. 2022. Searching in clutter: Visual behavior and performance of expert action video game players. Applied Ergonomics 99 (2022) 103628.","DOI":"10.1016\/j.apergo.2021.103628"},{"key":"e_1_3_3_2_6_1","unstructured":"Aaron Hurst Adam Lerer Adam\u00a0P Goucher Adam Perelman Aditya Ramesh Aidan Clark AJ Ostrow Akila Welihinda Alan Hayes Alec Radford et\u00a0al. 2024. Gpt-4o system card. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2410.21276 (2024)."},{"key":"e_1_3_3_2_7_1","doi-asserted-by":"crossref","unstructured":"Matej Morav\u010d\u00edk Martin Schmid Neil Burch Viliam Lis\u1ef3 Dustin Morrill Nolan Bard Trevor Davis Kevin Waugh Michael Johanson and Michael Bowling. 2017. Deepstack: Expert-level artificial intelligence in heads-up no-limit poker. Science 356 6337 (2017) 508\u2013513.","DOI":"10.1126\/science.aam6960"},{"key":"e_1_3_3_2_8_1","doi-asserted-by":"crossref","unstructured":"Stephen\u00a0E. Palmer. 2003. Visual Perception of Objects. Handbook of Psychology (2003) 177\u2013211.","DOI":"10.1002\/0471264385.wei0407"},{"key":"e_1_3_3_2_9_1","doi-asserted-by":"crossref","unstructured":"David Silver Thomas Hubert Julian Schrittwieser Ioannis Antonoglou Matthew Lai Arthur Guez Marc Lanctot Laurent Sifre Dharshan Kumaran Thore Graepel et\u00a0al. 2018. A general reinforcement learning algorithm that masters chess shogi and Go through self-play. Science 362 6419 (2018) 1140\u20131144.","DOI":"10.1126\/science.aar6404"},{"key":"e_1_3_3_2_10_1","unstructured":"Gemini Team Rohan Anil Sebastian Borgeaud Jean-Baptiste Alayrac Jiahui Yu Radu Soricut Johan Schalkwyk Andrew\u00a0M Dai Anja Hauth Katie Millican et\u00a0al. 2023. Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2312.11805 (2023)."},{"key":"e_1_3_3_2_11_1","unstructured":"Xinyu Wang Bohan Zhuang and Qi Wu. 2025. Are large vision language models good game players?arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.02358 (2025)."},{"key":"e_1_3_3_2_12_1","unstructured":"Jianwei Yang Hao Zhang Feng Li Xueyan Zou Chunyuan Li and Jianfeng Gao. 2023. Set-of-mark prompting unleashes extraordinary visual grounding in gpt-4v. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2310.11441 (2023)."},{"key":"e_1_3_3_2_13_1","unstructured":"Haoran Zhang Hangyu Guo Shuyue Guo Meng Cao Wenhao Huang Jiaheng Liu and Ge Zhang. 2024. Ing-vp: Mllms cannot play easy vision-based games yet. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2410.06555 (2024)."},{"key":"e_1_3_3_2_14_1","unstructured":"Xiangxi Zheng Linjie Li Zhengyuan Yang Ping Yu Alex\u00a0Jinpeng Wang Rui Yan Yuan Yao and Lijuan Wang. 2025. V-MAGE: A Game Evaluation Framework for Assessing Vision-Centric Capabilities in Multimodal Large Language Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2504.06148 (2025)."}],"event":{"name":"SA Technical Communications '25: SIGGRAPH Asia 2025 Technical Communications","location":"Hong Kong Hong Kong","acronym":"SA Technical Communications '25","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["Proceedings of the SIGGRAPH Asia 2025 Technical Communications"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3757376.3771402","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T09:09:50Z","timestamp":1765357790000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3757376.3771402"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,14]]},"references-count":13,"alternative-id":["10.1145\/3757376.3771402","10.1145\/3757376"],"URL":"https:\/\/doi.org\/10.1145\/3757376.3771402","relation":{},"subject":[],"published":{"date-parts":[[2025,12,14]]},"assertion":[{"value":"2025-12-14","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}