{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:15:34Z","timestamp":1765340134403,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":43,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62476097"],"award-info":[{"award-number":["62476097"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Fundamental Research Funds for the Central Universities, South China University of Technology","award":["x2rjD2250190"],"award-info":[{"award-number":["x2rjD2250190"]}]},{"name":"Guangdong Provincial Fund for Basic and Applied Basic Research?Regional Joint Fund Project (Key Project)","award":["2023B1515120078"],"award-info":[{"award-number":["2023B1515120078"]}]},{"name":"Guangdong Provincial Natural Science Foundation for Outstanding Youth Team Project","award":["2024B1515040010"],"award-info":[{"award-number":["2024B1515040010"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755425","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T06:47:18Z","timestamp":1761374838000},"page":"1754-1763","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["From Model Diagram to Code: A Benchmark Dataset and Multi-Agent Framework"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4086-2743","authenticated-orcid":false,"given":"Mengzhen","family":"Wang","sequence":"first","affiliation":[{"name":"South China University of Technology, Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-4168-5854","authenticated-orcid":false,"given":"Xunbin","family":"Huang","sequence":"additional","affiliation":[{"name":"South China University of Technology, GuangZhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6833-7879","authenticated-orcid":false,"given":"Jiayuan","family":"Xie","sequence":"additional","affiliation":[{"name":"Hong Kong Polytechnic University, Hong Kong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-0645-1650","authenticated-orcid":false,"given":"Shukai","family":"Ma","sequence":"additional","affiliation":[{"name":"South China University of Technology, Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-7988-6459","authenticated-orcid":false,"given":"Jiale","family":"Men","sequence":"additional","affiliation":[{"name":"South China University of Technology, Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-1075-8652","authenticated-orcid":false,"given":"Dayong","family":"Liang","sequence":"additional","affiliation":[{"name":"South China University of Technology, Guangzhu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1767-789X","authenticated-orcid":false,"given":"Yi","family":"Cai","sequence":"additional","affiliation":[{"name":"South China University of Technology, Guangzhou, China and Key Laboratory of Big Data and Intelligent Robot (SCUT), Ministry of Education, Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al.","author":"Achiam Josh","year":"2023","unstructured":"Josh Achiam, Steven Adler, Sandhini Agarwal, Lama Ahmad, Ilge Akkaya, Florencia Leoni Aleman, Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al., 2023. Gpt-4 technical report. arXiv preprint arXiv:2303.08774 (2023)."},{"key":"e_1_3_2_1_2_1","unstructured":"Mistral AI. 2023. Announcing Mistral 7B. https:\/\/www.mistral.ai\/news\/announcing-mistral-7b. Accessed: 2024-03-30."},{"key":"e_1_3_2_1_3_1","unstructured":"Amazon Web Services. 2024. Amazon Nova Lite v1 Documentation. https:\/\/docs.aws.amazon.com\/zh_cn\/nova\/latest\/userguide\/what-is-nova.html. Accessed: 2025-03-30."},{"key":"e_1_3_2_1_4_1","unstructured":"AI Anthropic. 2024. Introducing the next generation of claude."},{"key":"e_1_3_2_1_5_1","unstructured":"Shuai Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge Sibo Song Kai Dang Peng Wang Shijie Wang Jun Tang et al. 2025. Qwen2. 5-vl technical report. arXiv preprint arXiv:2502.13923 (2025)."},{"key":"e_1_3_2_1_6_1","first-page":"2296","volume-title":"Proceedings of the 2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation (LREC-COLING","author":"Cao Liuwen","year":"2024","unstructured":"Liuwen Cao, Yi Cai, Jiexin Wang, Hongkui He, and Hailin Huang. 2024. Beyond code: Evaluate thought steps for complex code generation. In Proceedings of the 2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation (LREC-COLING 2024). 2296-2306."},{"key":"e_1_3_2_1_7_1","volume-title":"Proceedings of the 31st International Conference on Computational Linguistics. 3043-3056","author":"Cao Liuwen","year":"2025","unstructured":"Liuwen Cao, Hongkui He, Hailin Huang, Jiexin Wang, and Yi Cai. 2025. Rethinking-based Code Summarization with Chain of Comments. In Proceedings of the 31st International Conference on Computational Linguistics. 3043-3056."},{"key":"e_1_3_2_1_8_1","volume-title":"Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, et al.","author":"Chen Mark","year":"2021","unstructured":"Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde De Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, et al., 2021. Evaluating large language models trained on code. arXiv preprint arXiv:2107.03374 (2021)."},{"key":"e_1_3_2_1_9_1","first-page":"4171","volume-title":"Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: human language technologies","volume":"1","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. Bert: Pre-training of deep bidirectional transformers for language understanding. In Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: human language technologies, volume 1 (long and short papers). 4171-4186."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3641554.3701782"},{"key":"e_1_3_2_1_11_1","volume-title":"Chatglm: A family of large language models from glm-130b to glm-4 all tools. arXiv preprint arXiv:2406.12793","author":"Aohan Zeng Team GLM","year":"2024","unstructured":"Team GLM, Aohan Zeng, Bin Xu, Bowen Wang, Chenhui Zhang, Da Yin, Dan Zhang, Diego Rojas, Guanyu Feng, Hanlin Zhao, et al., 2024. Chatglm: A family of large language models from glm-130b to glm-4 all tools. arXiv preprint arXiv:2406.12793 (2024)."},{"key":"e_1_3_2_1_12_1","volume-title":"Chartllama: A multimodal llm for chart understanding and generation. arXiv preprint arXiv:2311.16483","author":"Han Yucheng","year":"2023","unstructured":"Yucheng Han, Chi Zhang, Xin Chen, Xu Yang, Zhibin Wang, Gang Yu, Bin Fu, and Hanwang Zhang. 2023. Chartllama: A multimodal llm for chart understanding and generation. arXiv preprint arXiv:2311.16483 (2023)."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.211"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681294"},{"key":"e_1_3_2_1_15_1","unstructured":"Binyuan Hui Jian Yang Zeyu Cui Jiaxi Yang Dayiheng Liu Lei Zhang Tianyu Liu Jiajun Zhang Bowen Yu Keming Lu et al. 2024. Qwen2. 5-coder technical report. arXiv preprint arXiv:2409.12186 (2024)."},{"key":"e_1_3_2_1_16_1","first-page":"2011","article-title":"Systems and software engineering-architecture description","volume":"42010","author":"IEEE.","year":"2011","unstructured":"ISO\/IEC\/IEEE. 2011. Systems and software engineering-architecture description. ISO\/IEC\/IEEE 42010: 2011 (E)(Revision of ISO\/IEC 42010: 2007 and IEEE Std 1471-2000), Vol. 2011 (2011), 1-46.","journal-title":"ISO\/IEC\/IEEE"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642822"},{"key":"e_1_3_2_1_18_1","unstructured":"Aonian Li Bangwei Gong Bo Yang Boji Shan Chang Liu Cheng Zhu Chunhao Zhang Congchao Guo Da Chen Dong Li et al. 2025. Minimax-01: Scaling foundation models with lightning attention. arXiv preprint arXiv:2501.08313 (2025)."},{"key":"e_1_3_2_1_19_1","volume-title":"Llava-next-interleave: Tackling multi-image, video, and 3d in large multimodal models. arXiv preprint arXiv:2407.07895","author":"Li Feng","year":"2024","unstructured":"Feng Li, Renrui Zhang, Hao Zhang, Yuanhan Zhang, Bo Li, Wei Li, Zejun Ma, and Chunyuan Li. 2024c. Llava-next-interleave: Tackling multi-image, video, and 3d in large multimodal models. arXiv preprint arXiv:2407.07895 (2024)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.42"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.775"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-022-01639-z"},{"key":"e_1_3_2_1_23_1","volume-title":"Visual instruction tuning. Advances in neural information processing systems","author":"Liu Haotian","year":"2023","unstructured":"Haotian Liu, Chunyuan Li, Qingyang Wu, and Yong Jae Lee. 2023. Visual instruction tuning. Advances in neural information processing systems, Vol. 36 (2023), 34892-34916."},{"key":"e_1_3_2_1_24_1","unstructured":"Haoyu Lu Wen Liu Bo Zhang Bingxuan Wang Kai Dong Bo Liu Jingxiang Sun Tongzheng Ren Zhuoshu Li Hao Yang et al. 2024. Deepseek-vl: towards real-world vision-language understanding. arXiv preprint arXiv:2403.05525 (2024)."},{"key":"e_1_3_2_1_25_1","unstructured":"Meta AI. 2024. Llama 3.2 Model Card and Prompt Formats. https:\/\/www.llama.com\/docs\/model-cards-and-prompt-formats\/llama3_2\/. Accessed: 2025-03-30."},{"volume-title":"Benchmarking Generative Models on Computational Thinking Tests in Elementary Visual Programming. In The Thirty-eight Conference on Neural Information Processing Systems Datasets and Benchmarks Track.","author":"P\u0103durean Victor-Alexandru","key":"e_1_3_2_1_26_1","unstructured":"Victor-Alexandru P\u0103durean and Adish Singla. [n.d.]. Benchmarking Generative Models on Computational Thinking Tests in Elementary Visual Programming. In The Thirty-eight Conference on Neural Information Processing Systems Datasets and Benchmarks Track."},{"key":"e_1_3_2_1_27_1","volume-title":"International conference on machine learning. PmLR, 8748-8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al., 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PmLR, 8748-8763."},{"key":"e_1_3_2_1_28_1","first-page":"115058","article-title":"Image2Struct: Benchmarking Structure Extraction for Vision-Language Models","volume":"37","author":"Roberts Josselin","year":"2025","unstructured":"Josselin Roberts, Tony Lee, Chi Heem Wong, Michihiro Yasunaga, Yifan Mai, and Percy S Liang. 2025. Image2Struct: Benchmarking Structure Extraction for Vision-Language Models. Advances in Neural Information Processing Systems, Vol. 37 (2025), 115058-115097.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_29_1","volume-title":"Chartmimic: Evaluating lmm's cross-modal reasoning capability via chart-to-code generation. arXiv preprint arXiv:2406.09961","author":"Shi Chufan","year":"2024","unstructured":"Chufan Shi, Cheng Yang, Yaxin Liu, Bo Shui, Junjie Wang, Mohan Jing, Linran Xu, Xinyu Zhu, Siheng Li, Yuxiang Zhang, et al., 2024. Chartmimic: Evaluating lmm's cross-modal reasoning capability via chart-to-code generation. arXiv preprint arXiv:2406.09961 (2024)."},{"key":"e_1_3_2_1_30_1","volume-title":"Design2code: How far are we from automating front-end engineering? arXiv e-prints","author":"Si Chenglei","year":"2024","unstructured":"Chenglei Si, Yanzhe Zhang, Zhengyuan Yang, Ruibo Liu, and Diyi Yang. 2024. Design2code: How far are we from automating front-end engineering? arXiv e-prints (2024), arXiv-2403."},{"key":"e_1_3_2_1_31_1","volume-title":"Learning UI-to-Code Reverse Generator Using Visual Critic Without Rendering. arXiv preprint arXiv:2305.14637","author":"Soselia Davit","year":"2023","unstructured":"Davit Soselia, Khalid Saifullah, and Tianyi Zhou. 2023. Learning UI-to-Code Reverse Generator Using Visual Critic Without Rendering. arXiv preprint arXiv:2305.14637 (2023)."},{"key":"e_1_3_2_1_32_1","unstructured":"Gemini Team Rohan Anil Sebastian Borgeaud Jean-Baptiste Alayrac Jiahui Yu Radu Soricut Johan Schalkwyk Andrew M Dai Anja Hauth Katie Millican et al. 2023. Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:2312.11805 (2023)."},{"key":"e_1_3_2_1_33_1","unstructured":"Gemma Team Aishwarya Kamath Johan Ferret Shreya Pathak Nino Vieillard Ramona Merhej Sarah Perrin Tatiana Matejovicova Alexandre Ram\u00e9 Morgane Rivi\u00e8re et al. 2025. Gemma 3 Technical Report. arXiv preprint arXiv:2503.19786 (2025)."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626252.3630916"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.mechatronics.2014.05.003"},{"key":"e_1_3_2_1_36_1","volume-title":"Automatically generating UI code from screenshot: A divide-and-conquer-based approach. arXiv preprint arXiv:2406.16386","author":"Wan Yuxuan","year":"2024","unstructured":"Yuxuan Wan, Chaozheng Wang, Yi Dong, Wenxuan Wang, Shuqing Li, Yintong Huo, and Michael R Lyu. 2024. Automatically generating UI code from screenshot: A divide-and-conquer-based approach. arXiv preprint arXiv:2406.16386 (2024)."},{"key":"e_1_3_2_1_37_1","unstructured":"Peng Wang Shuai Bai Sinan Tan Shijie Wang Zhihao Fan Jinze Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge et al. 2024. Qwen2-vl: Enhancing vision-language model's perception of the world at any resolution. arXiv preprint arXiv:2409.12191 (2024)."},{"key":"e_1_3_2_1_38_1","volume-title":"Plot2code: A comprehensive benchmark for evaluating multi-modal large language models in code generation from scientific plots. arXiv preprint arXiv:2405.07990","author":"Wu Chengyue","year":"2024","unstructured":"Chengyue Wu, Yixiao Ge, Qiushan Guo, Jiahao Wang, Zhixuan Liang, Zeyu Lu, Ying Shan, and Ping Luo. 2024. Plot2code: A comprehensive benchmark for evaluating multi-modal large language models in code generation from scientific plots. arXiv preprint arXiv:2405.07990 (2024)."},{"key":"e_1_3_2_1_39_1","volume-title":"Yi: Open foundation models by 01. ai. arXiv preprint arXiv:2403.04652","author":"Young Alex","year":"2024","unstructured":"Alex Young, Bei Chen, Chao Li, Chengen Huang, Ge Zhang, Guanwei Zhang, Guoyin Wang, Heng Li, Jiangcheng Zhu, Jianqun Chen, et al., 2024. Yi: Open foundation models by 01. ai. arXiv preprint arXiv:2403.04652 (2024)."},{"key":"e_1_3_2_1_40_1","volume-title":"HumanEval-V: Evaluating Visual Understanding and Reasoning Abilities of Large Multimodal Models Through Coding Tasks. arXiv preprint arXiv:2410.12381","author":"Zhang Fengji","year":"2024","unstructured":"Fengji Zhang, Linquan Wu, Huiyu Bai, Guancheng Lin, Xiao Li, Xiao Yu, Yue Wang, Bei Chen, and Jacky Keung. 2024b. HumanEval-V: Evaluating Visual Understanding and Reasoning Abilities of Large Multimodal Models Through Coding Tasks. arXiv preprint arXiv:2410.12381 (2024)."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.485"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3543507.3583457"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3580305.3599790"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755425","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:12:38Z","timestamp":1765339958000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755425"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":43,"alternative-id":["10.1145\/3746027.3755425","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755425","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}