{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,12]],"date-time":"2025-12-12T13:50:29Z","timestamp":1765547429296,"version":"3.41.0"},"publisher-location":"Cham","reference-count":29,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031923869","type":"print"},{"value":"9783031923876","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-92387-6_26","type":"book-chapter","created":{"date-parts":[[2025,5,25]],"date-time":"2025-05-25T18:42:41Z","timestamp":1748198561000},"page":"374-389","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Level Up Your Tutorials: VLMs for\u00a0Game Tutorials Quality Assessment"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5067-2118","authenticated-orcid":false,"given":"Daniele","family":"Rege Cambrin","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-2462-8778","authenticated-orcid":false,"given":"Gabriele","family":"Scaffidi Militone","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2911-4522","authenticated-orcid":false,"given":"Luca","family":"Colomba","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6798-2761","authenticated-orcid":false,"given":"Giovanni","family":"Malnati","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0538-9775","authenticated-orcid":false,"given":"Daniele","family":"Apiletti","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1263-7522","authenticated-orcid":false,"given":"Paolo","family":"Garza","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,5,12]]},"reference":[{"key":"26_CR1","unstructured":"Abdin, M., et al.: Phi-3 technical report: a highly capable language model locally on your phone (2024). https:\/\/arxiv.org\/abs\/2404.14219"},{"key":"26_CR2","unstructured":"AI@Meta: Llama 3 model card (2024). https:\/\/github.com\/meta-llama\/llama3\/blob\/main\/MODEL_CARD.md"},{"key":"26_CR3","first-page":"23716","volume":"35","author":"JB Alayrac","year":"2022","unstructured":"Alayrac, J.B., et al.: Flamingo: a visual language model for few-shot learning. Adv. Neural. Inf. Process. Syst. 35, 23716\u201323736 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"26_CR4","doi-asserted-by":"publisher","unstructured":"Ariyurek, S., Betin-Can, A., Surer, E.: Automated video game testing using synthetic and humanlike agents. IEEE Trans. Games 13(1), 50\u201367 (2021). https:\/\/doi.org\/10.1109\/TG.2019.2947597","DOI":"10.1109\/TG.2019.2947597"},{"issue":"3","key":"26_CR5","doi-asserted-by":"publisher","first-page":"240","DOI":"10.1177\/1527476414525241","volume":"16","author":"E Bulut","year":"2015","unstructured":"Bulut, E.: Playboring in the tester pit: the convergence of precarity and the degradation of fun in video game testing. Telev. New Media 16(3), 240\u2013258 (2015). https:\/\/doi.org\/10.1177\/1527476414525241","journal-title":"Telev. New Media"},{"key":"26_CR6","doi-asserted-by":"publisher","unstructured":"Cao, S., Liu, F.: Learning to play: understanding in-game tutorials with a pilot study on implicit tutorials. Heliyon 8(11), e11482 (2022). https:\/\/doi.org\/10.1016\/j.heliyon.2022.e11482","DOI":"10.1016\/j.heliyon.2022.e11482"},{"key":"26_CR7","unstructured":"Chen, K., Thapa, R., Chalamala, R., Athiwaratkun, B., Song, S.L., Zou, J.: Dragonfly: multi-resolution zoom supercharges large visual-language model (2024)"},{"key":"26_CR8","doi-asserted-by":"crossref","unstructured":"Chen, Z., et al.: How far are we TO GPT-4V? closing the gap to commercial multimodal models with open-source suites (2024). https:\/\/arxiv.org\/abs\/2404.16821","DOI":"10.1007\/s11432-024-4231-5"},{"key":"26_CR9","doi-asserted-by":"crossref","unstructured":"Chen, Z., et al.: Internvl: scaling up vision foundation models and aligning for generic visual-linguistic tasks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 24185\u201324198 (2024)","DOI":"10.1109\/CVPR52733.2024.02283"},{"key":"26_CR10","unstructured":"OpenCompass Contributors: Opencompass: a universal evaluation platform for foundation models (2023). https:\/\/github.com\/open-compass\/opencompass"},{"key":"26_CR11","doi-asserted-by":"crossref","unstructured":"Hong, W., et al.: Cogagent: a visual language model for GUI agents. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 14281\u201314290 (2024). https:\/\/openaccess.thecvf.com\/content\/CVPR2024\/html\/Hong_CogAgent_A_Visual_Language_Model_for_GUI_Agents_CVPR_2024_paper.html","DOI":"10.1109\/CVPR52733.2024.01354"},{"key":"26_CR12","unstructured":"Jia, C., et al.: Scaling up visual and vision-language representation learning with noisy text supervision. In: Meila, M., Zhang, T. (eds.) Proceedings of the 38th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a0139, pp. 4904\u20134916. PMLR (2021). https:\/\/proceedings.mlr.press\/v139\/jia21b.html"},{"key":"26_CR13","unstructured":"Khorikov, V.: Unit Testing Principles, Practices, and Patterns. Simon and Schuster (2020)"},{"issue":"5","key":"26_CR14","first-page":"713","volume":"17","author":"PS Liao","year":"2001","unstructured":"Liao, P.S., Chen, T.S., Chung, P.C.: A fast algorithm for multilevel thresholding. J. Inf. Sci. Eng. 17(5), 713\u2013727 (2001)","journal-title":"J. Inf. Sci. Eng."},{"key":"26_CR15","unstructured":"Lin, C.Y.: ROUGE: a package for automatic evaluation of summaries. In: Text Summarization Branches Out, Barcelona, Spain, pp. 74\u201381. Association for Computational Linguistics (2004). https:\/\/aclanthology.org\/W04-1013"},{"issue":"6","key":"26_CR16","doi-asserted-by":"publisher","first-page":"4006","DOI":"10.1007\/s10664-019-09733-6","volume":"24","author":"D Lin","year":"2019","unstructured":"Lin, D., Bezemer, C.-P., Hassan, A.E.: Identifying gameplay videos that exhibit bugs in computer games. Empir. Softw. Eng. 24(6), 4006\u20134033 (2019). https:\/\/doi.org\/10.1007\/s10664-019-09733-6","journal-title":"Empir. Softw. Eng."},{"key":"26_CR17","unstructured":"Liu, H., Li, C., Wu, Q., Lee, Y.J.: Visual instruction tuning. In: Advances in Neural Information Processing Systems, vol. 36 (2024)"},{"key":"26_CR18","doi-asserted-by":"crossref","unstructured":"Liu, Y., et al.: Mmbench: is your multi-modal model an all-around player? arXiv:2307.06281 (2023)","DOI":"10.1007\/978-3-031-72658-3_13"},{"key":"26_CR19","unstructured":"Liu, Z., et al.: Vision-driven Automated Mobile GUI Testing via Multimodal Large Language Model (2024). http:\/\/arxiv.org\/abs\/2407.03037. arXiv:2407.03037"},{"key":"26_CR20","unstructured":"OpenAI: GPT-4 technical report (2024). https:\/\/arxiv.org\/abs\/2303.08774"},{"key":"26_CR21","doi-asserted-by":"publisher","unstructured":"Paduraru, C., Paduraru, M., Stefanescu, A.: RiverGame - a game testing tool using artificial intelligence. In: 2022 IEEE Conference on Software Testing, Verification and Validation (ICST), pp. 422\u2013432 (2022). https:\/\/doi.org\/10.1109\/ICST53961.2022.00048. https:\/\/ieeexplore.ieee.org\/document\/9787838. ISSN: 2159-4848","DOI":"10.1109\/ICST53961.2022.00048"},{"key":"26_CR22","unstructured":"Peng, Z., et al.: Kosmos-2: grounding multimodal large language models to the world. arXiv preprint arXiv:2306.14824 (2023)"},{"key":"26_CR23","doi-asserted-by":"publisher","unstructured":"Politowski, C., Petrillo, F., Gu\u00e9h\u00e9neuc, Y.G.: A survey of video game testing. In: 2021 IEEE\/ACM International Conference on Automation of Software Test (AST), pp. 90\u201399 (2021). https:\/\/doi.org\/10.1109\/AST52587.2021.00018. https:\/\/ieeexplore.ieee.org\/abstract\/document\/9463010","DOI":"10.1109\/AST52587.2021.00018"},{"key":"26_CR24","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR (2021)"},{"key":"26_CR25","doi-asserted-by":"crossref","unstructured":"Reimers, N., Gurevych, I.: Sentence-bert: sentence embeddings using siamese bert-networks. In: Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing. Association for Computational Linguistics (2019). https:\/\/arxiv.org\/abs\/1908.10084","DOI":"10.18653\/v1\/D19-1410"},{"key":"26_CR26","doi-asserted-by":"crossref","unstructured":"Taesiri, M.R., Feng, T., Bezemer, C.P., Nguyen, A.: Glitchbench: can large multimodal models detect video game glitches? In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 22444\u201322455 (2024). https:\/\/openaccess.thecvf.com\/content\/CVPR2024\/html\/Taesiri_GlitchBench_Can_Large_Multimodal_Models_Detect_Video_Game_Glitches_CVPR_2024_paper.html","DOI":"10.1109\/CVPR52733.2024.02118"},{"key":"26_CR27","doi-asserted-by":"publisher","unstructured":"Taesiri, M.R., Macklon, F., Wang, Y., Shen, H., Bezemer, C.P.: Large Language Models are Pretty Good Zero-Shot Video Game Bug Detectors (2022). https:\/\/doi.org\/10.48550\/arXiv.2210.02506. http:\/\/arxiv.org\/abs\/2210.02506. arXiv:2210.02506","DOI":"10.48550\/arXiv.2210.02506"},{"key":"26_CR28","doi-asserted-by":"crossref","unstructured":"Yue, X., et al.: Mmmu: a massive multi-discipline multimodal understanding and reasoning benchmark for expert AGI. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 9556\u20139567 (2024)","DOI":"10.1109\/CVPR52733.2024.00913"},{"key":"26_CR29","unstructured":"Zhang, T., Kishore, V., Wu, F., Weinberger, K.Q., Artzi, Y.: Bertscore: evaluating text generation with bert. In: International Conference on Learning Representations (2020). https:\/\/openreview.net\/forum?id=SkeHuCVFDr"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024 Workshops"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-92387-6_26","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,25]],"date-time":"2025-05-25T18:42:50Z","timestamp":1748198570000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-92387-6_26"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9783031923869","9783031923876"],"references-count":29,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-92387-6_26","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"12 May 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}