{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T07:15:43Z","timestamp":1779174943561,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":46,"publisher":"ACM","funder":[{"name":"The Research Council of Finland","award":["358726"],"award-info":[{"award-number":["358726"]}]},{"name":"The Research Council of Finland","award":["353267"],"award-info":[{"award-number":["353267"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,3,23]]},"DOI":"10.1145\/3789624.3789641","type":"proceedings-article","created":{"date-parts":[[2026,3,20]],"date-time":"2026-03-20T11:09:18Z","timestamp":1774004958000},"page":"1-16","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Leveraging Large Language Models for Twitch Video Clip Understanding and Summarization"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2275-3866","authenticated-orcid":false,"given":"Jari","family":"Lindroos","sequence":"first","affiliation":[{"name":"University of Jyv\u00e4skyl\u00e4, Jyv\u00e4skyl\u00e4, Finland"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1492-4074","authenticated-orcid":false,"given":"Raine","family":"Koskimaa","sequence":"additional","affiliation":[{"name":"University of Jyv\u00e4skyl\u00e4, Jyv\u00e4skyl\u00e4, Finland"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3485-8585","authenticated-orcid":false,"given":"Jaakko","family":"Peltonen","sequence":"additional","affiliation":[{"name":"Tampere University, Tampere, Finland"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8678-4683","authenticated-orcid":false,"given":"Tanja","family":"V\u00e4lisalo","sequence":"additional","affiliation":[{"name":"University of Jyv\u00e4skyl\u00e4, Jyv\u00e4skyl\u00e4, Finland"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0440-7337","authenticated-orcid":false,"given":"Ida","family":"Toivanen","sequence":"additional","affiliation":[{"name":"University of Jyv\u00e4skyl\u00e4, Jyv\u00e4skyl\u00e4, Finland"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2026,3,22]]},"reference":[{"key":"e_1_3_3_2_2_2","doi-asserted-by":"crossref","unstructured":"Toqa Alaa Ahmad Mongy Assem Bakr Mariam Diab and Walid Gomaa. 2024. Video Summarization Techniques: A Comprehensive Review. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2410.04449 (2024).","DOI":"10.5220\/0012936400003822"},{"key":"e_1_3_3_2_3_2","unstructured":"Mario Barbara and Alaa Maalouf. 2025. Prompts to Summaries: Zero-Shot Language-Guided Video Summarization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2506.10807 (2025)."},{"key":"e_1_3_3_2_4_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICACRS62842.2024.10841652"},{"key":"e_1_3_3_2_5_2","doi-asserted-by":"crossref","unstructured":"Yanran Chen and Steffen Eger. 2023. Menli: Robust evaluation metrics from natural language inference. Transactions of the Association for Computational Linguistics 11 (2023) 804\u2013825.","DOI":"10.1162\/tacl_a_00576"},{"key":"e_1_3_3_2_6_2","volume-title":"Forty-first International Conference on Machine Learning","author":"Chiang Wei-Lin","year":"2024","unstructured":"Wei-Lin Chiang, Lianmin Zheng, Ying Sheng, Anastasios\u00a0Nikolas Angelopoulos, Tianle Li, Dacheng Li, Banghua Zhu, Hao Zhang, Michael Jordan, Joseph\u00a0E Gonzalez, et\u00a0al. 2024. Chatbot arena: An open platform for evaluating llms by human preference. In Forty-first International Conference on Machine Learning."},{"key":"e_1_3_3_2_7_2","unstructured":"Gheorghe Comanici Eric Bieber Mike Schaekermann Ice Pasupat Noveen Sachdeva Inderjit Dhillon Marcel Blistein Ori Ram Dan Zhang Evan Rosen et\u00a0al. 2025. Gemini 2.5: Pushing the frontier with advanced reasoning multimodality long context and next generation agentic capabilities. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2507.06261 (2025)."},{"key":"e_1_3_3_2_8_2","doi-asserted-by":"publisher","DOI":"10.1145\/3027063.3052765"},{"key":"e_1_3_3_2_9_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02245"},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"crossref","unstructured":"Roberto Gallotta Graham Todd Marvin Zammit Sam Earle Antonios Liapis Julian Togelius and Georgios\u00a0N Yannakakis. 2024. Large language models and games: A survey and roadmap. IEEE Transactions on Games (2024).","DOI":"10.1109\/TG.2024.3461510"},{"key":"e_1_3_3_2_11_2","unstructured":"Mingqi Gao Jie Ruan Renliang Sun Xunjian Yin Shiping Yang and Xiaojun Wan. 2023. Human-like summarization evaluation with chatgpt. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2304.02554 (2023)."},{"key":"e_1_3_3_2_12_2","volume-title":"Gemini 3 Pro Model Card","author":"DeepMind Gemini Team, Google","year":"2025","unstructured":"Gemini Team, Google DeepMind. 2025. Gemini 3 Pro Model Card. Model Card. Google DeepMind. https:\/\/storage.googleapis.com\/deepmind-media\/Model-Cards\/Gemini-3-Pro-Model-Card.pdf"},{"key":"e_1_3_3_2_13_2","unstructured":"Ch Gough. 2022. eSports audience size worldwide from 2020 to 2025. Statista. Retrieved December 2025 29 (2022) 2022."},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"crossref","unstructured":"Deeksha Gupta and Akashdeep Sharma. 2023. A comprehensive study of automatic video summarization techniques. Artificial Intelligence Review 56 10 (2023) 11473\u201311633.","DOI":"10.1007\/s10462-023-10429-z"},{"key":"e_1_3_3_2_15_2","doi-asserted-by":"crossref","unstructured":"Mika H\u00e4m\u00e4l\u00e4inen Jack Rueter and Khalid Alnajjar. 2024. Analyzing Pok\\ \u2019emon and Mario Streamers\u2019 Twitch Chat with LLM-based User Embeddings. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2411.10934 (2024).","DOI":"10.18653\/v1\/2024.nlp4dh-1.48"},{"key":"e_1_3_3_2_16_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01723"},{"key":"e_1_3_3_2_17_2","doi-asserted-by":"publisher","DOI":"10.1145\/3706598.3713880"},{"key":"e_1_3_3_2_18_2","doi-asserted-by":"crossref","unstructured":"Harper Kohls Jacob\u00a0L Hiler and Laurel\u00a0Aynne Cook. 2023. Why do we twitch? Vicarious consumption in video-game livestreaming. Journal of Consumer Marketing 40 6 (2023) 639\u2013650.","DOI":"10.1108\/JCM-03-2020-3727"},{"key":"e_1_3_3_2_19_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01768"},{"key":"e_1_3_3_2_20_2","unstructured":"Shicheng Li Lei Li Kun Ouyang Shuhuai Ren Yuanxin Liu Yuanxing Zhang Fuzheng Zhang Lingpeng Kong Qi Liu and Xu Sun. 2025. TEMPLE: Temporal Preference Learning of Video LLMs via Difficulty Scheduling and Pre-SFT Alignment. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.16929 (2025)."},{"key":"e_1_3_3_2_21_2","unstructured":"Yanjun Li Yuqian Fu Tianwen Qian Qi\u2019ao Xu Silong Dai Danda\u00a0Pani Paudel Luc Van\u00a0Gool and Xiaoling Wang. 2025. EgoCross: Benchmarking Multimodal Large Language Models for Cross-Domain Egocentric Video Question Answering. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2508.10729 (2025)."},{"key":"e_1_3_3_2_22_2","unstructured":"Zhong-Zhi Li Duzhen Zhang Ming-Liang Zhang Jiaxin Zhang Zengyan Liu Yuxuan Yao Haotian Xu Junhao Zheng Pei-Jie Wang Xiuyi Chen et\u00a0al. 2025. From system 1 to system 2: A survey of reasoning large language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2502.17419 (2025)."},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"publisher","DOI":"10.24251\/HICSS.2025.308"},{"key":"e_1_3_3_2_24_2","unstructured":"Wentao Lu Alexander Senchenko Abram Hindle and Cor-Paul Bezemer. 2025. Automated Bug Frame Retrieval from Gameplay Videos Using Vision-Language Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2508.04895 (2025)."},{"key":"e_1_3_3_2_25_2","unstructured":"Muhammad Maaz Hanoona Rasheed Salman Khan and Fahad\u00a0Shahbaz Khan. 2023. Video-chatgpt: Towards detailed video understanding via large vision and language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2306.05424 (2023)."},{"key":"e_1_3_3_2_26_2","unstructured":"David Melhart Matthew Barthet and Georgios\u00a0N Yannakakis. 2025. Can Large Language Models Capture Video Game Engagement? arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2502.04379 (2025)."},{"key":"e_1_3_3_2_27_2","doi-asserted-by":"publisher","DOI":"10.1145\/3589334.3645677"},{"key":"e_1_3_3_2_28_2","unstructured":"Kuan-Chen Mu Zhi-Yi Chin and Wei-Chen Chiu. 2024. Realizing Video Summarization from the Path of Language-based Semantic Understanding. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2410.04511 (2024)."},{"key":"e_1_3_3_2_29_2","unstructured":"Galann Pennec Zhengyuan Liu Nicholas Asher Philippe Muller and Nancy\u00a0F Chen. 2025. Integrating Video and Text: A Balanced Approach to Multimodal Summary Generation and Evaluation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2505.06594 (2025)."},{"key":"e_1_3_3_2_30_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.02252"},{"key":"e_1_3_3_2_31_2","unstructured":"Kanchana Ranasinghe Xiang Li Kumara Kahatapitiya and Michael\u00a0S Ryoo. 2024. Understanding long videos with multimodal language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2403.16998 (2024)."},{"key":"e_1_3_3_2_32_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01725"},{"key":"e_1_3_3_2_33_2","doi-asserted-by":"crossref","unstructured":"Jinhwan Sul Jihoon Han and Joonseok Lee. 2023. Mr. hisum: A large-scale dataset for video highlight detection and summarization. Advances in Neural Information Processing Systems 36 (2023) 40542\u201340555.","DOI":"10.52202\/075280-1764"},{"key":"e_1_3_3_2_34_2","doi-asserted-by":"publisher","DOI":"10.1145\/3640794.3665582"},{"key":"e_1_3_3_2_35_2","unstructured":"Mohammad\u00a0Reza Taesiri Abhijay Ghildyal Saman Zadtootaghaj Nabajeet Barman and Cor-Paul Bezemer. 2025. VideoGameQA-Bench: Evaluating Vision-Language Models for Video Game Quality Assurance. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2505.15952 (2025)."},{"key":"e_1_3_3_2_36_2","unstructured":"Yunlong Tang Jing Bi Siting Xu Luchuan Song Susan Liang Teng Wang Daoan Zhang Jie An Jingyang Lin Rongyi Zhu et\u00a0al. 2025. Video understanding with large language models: A survey. IEEE Transactions on Circuits and Systems for Video Technology (2025)."},{"key":"e_1_3_3_2_37_2","unstructured":"Gemini Team Petko Georgiev Ving\u00a0Ian Lei Ryan Burnell Libin Bai Anmol Gulati Garrett Tanzer Damien Vincent Zhufeng Pan Shibo Wang et\u00a0al. 2024. Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2403.05530 (2024)."},{"key":"e_1_3_3_2_38_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642839"},{"key":"e_1_3_3_2_39_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.02131"},{"key":"e_1_3_3_2_40_2","unstructured":"Yifei Wang Zhenkai Li Tianwen Qian Huanran Zheng Zheng Wang Yuqian Fu and Xiaoling Wang. 2025. StreamEQA: Towards Streaming Video Understanding for Embodied Scenarios. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2512.04451 (2025)."},{"key":"e_1_3_3_2_41_2","doi-asserted-by":"crossref","unstructured":"Jason Wei Xuezhi Wang Dale Schuurmans Maarten Bosma Fei Xia Ed Chi Quoc\u00a0V Le Denny Zhou et\u00a0al. 2022. Chain-of-thought prompting elicits reasoning in large language models. Advances in neural information processing systems 35 (2022) 24824\u201324837.","DOI":"10.52202\/068431-1800"},{"key":"e_1_3_3_2_42_2","unstructured":"Jianlong Wu Wei Liu Ye Liu Meng Liu Liqiang Nie Zhouchen Lin and Chang\u00a0Wen Chen. 2025. A survey on video temporal grounding with multimodal large language model. IEEE Transactions on Pattern Analysis and Machine Intelligence (2025)."},{"key":"e_1_3_3_2_43_2","doi-asserted-by":"publisher","DOI":"10.1145\/3778534.3778559"},{"key":"e_1_3_3_2_44_2","unstructured":"Zhihao Zhang Feiqi Cao Yingbin Mo Yiran Zhang Josiah Poon and Caren Han. 2024. Game-MUG: Multimodal oriented game situation understanding and commentary generation dataset. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2404.19175 (2024)."},{"key":"e_1_3_3_2_45_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.02025"},{"key":"e_1_3_3_2_46_2","unstructured":"Xu Zheng Zihao Dongfang Lutao Jiang Boyuan Zheng Yulong Guo Zhenquan Zhang Giuliano Albanese Runyi Yang Mengjiao Ma Zixin Zhang et\u00a0al. 2025. Multimodal Spatial Reasoning in the Large Model Era: A Survey and Benchmarks. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2510.25760 (2025)."},{"key":"e_1_3_3_2_47_2","doi-asserted-by":"crossref","unstructured":"Pengyuan Zhou Lin Wang Zhi Liu Yanbin Hao Pan Hui Sasu Tarkoma and Jussi Kangasharju. 2024. A survey on generative ai and llm for video generation understanding and streaming. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2404.16038 (2024).","DOI":"10.36227\/techrxiv.171172801.19993069\/v1"}],"event":{"name":"GamiFIN 2026: 10th Annual International GamiFIN Conference 2026","location":"Saariselk\u00e4 Finland","acronym":"GamiFIN 2026"},"container-title":["Proceedings of the 10th Annual International GamiFIN Conference 2026"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3789624.3789641","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T06:46:19Z","timestamp":1779173179000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3789624.3789641"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,22]]},"references-count":46,"alternative-id":["10.1145\/3789624.3789641","10.1145\/3789624"],"URL":"https:\/\/doi.org\/10.1145\/3789624.3789641","relation":{},"subject":[],"published":{"date-parts":[[2026,3,22]]},"assertion":[{"value":"2026-03-22","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}