{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T13:27:03Z","timestamp":1784381223213,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":63,"publisher":"ACM","funder":[{"name":"Scientific Research Innovation Capability Support Project for Young Faculty","award":["ZY-GXQNJSKYCXNLZCXM-I22"],"award-info":[{"award-number":["ZY-GXQNJSKYCXNLZCXM-I22"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755144","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:30:51Z","timestamp":1761377451000},"page":"3654-3663","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["Multi-Agent System for Comprehensive Soccer Understanding"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-9475-9193","authenticated-orcid":false,"given":"Jiayuan","family":"Rao","sequence":"first","affiliation":[{"name":"SAI, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-5625-9820","authenticated-orcid":false,"given":"Zifeng","family":"Li","sequence":"additional","affiliation":[{"name":"Zhiyuan College &amp; SAI, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-8717-338X","authenticated-orcid":false,"given":"Haoning","family":"Wu","sequence":"additional","affiliation":[{"name":"SAI, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5390-9053","authenticated-orcid":false,"given":"Ya","family":"Zhang","sequence":"additional","affiliation":[{"name":"SAI, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3196-2347","authenticated-orcid":false,"given":"Yanfeng","family":"Wang","sequence":"additional","affiliation":[{"name":"SAI, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3804-2639","authenticated-orcid":false,"given":"Weidi","family":"Xie","sequence":"additional","affiliation":[{"name":"SAI, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Anthropic. 2025. Claude 3.7 Sonnet. https:\/\/www.anthropic.com"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.279"},{"key":"e_1_3_2_1_3_1","volume-title":"arXiv preprint arXiv:2502.13923","author":"Bai Shuai","year":"2025","unstructured":"Shuai Bai, Keqin Chen, Xuejing Liu, Jialin Wang, Wenbin Ge, Sibo Song, Kai Dang, Peng Wang, Shijie Wang, Jun Tang, Humen Zhong, Yuanzhi Zhu, Mingkun Yang, Zhaohai Li, Jianqiang Wan, Pengfei Wang, Wei Ding, Zheren Fu, Yiheng Xu, Jiabo Ye, Xi Zhang, Tianbao Xie, Zesen Cheng, Hang Zhang, Zhibo Yang, Haiyang Xu, and Junyang Lin. 2025. Qwen2.5-VL Technical Report. arXiv preprint arXiv:2502.13923 (2025)."},{"key":"e_1_3_2_1_4_1","volume-title":"Microsoft coco captions: Data collection and evaluation server. arXiv preprint arXiv:1504.00325","author":"Chen Xinlei","year":"2015","unstructured":"Xinlei Chen, Hao Fang, Tsung-Yi Lin, Ramakrishna Vedantam, Saurabh Gupta, Piotr Doll\u00e1r, and C Lawrence Zitnick. 2015. Microsoft coco captions: Data collection and evaluation server. arXiv preprint arXiv:1504.00325 (2015)."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41597-022-01469-1"},{"key":"e_1_3_2_1_6_1","volume-title":"Xin Zhou, Karolina Seweryn, Mateusz Kowalczyk, et al.","author":"Cioppa Anthony","year":"2024","unstructured":"Anthony Cioppa, Silvio Giancola, Vladimir Somers, Victor Joos, Floriane Magera, Jan Held, Seyed Abolfazl Ghasemzadeh, Xin Zhou, Karolina Seweryn, Mateusz Kowalczyk, et al., 2024. SoccerNet 2024 Challenges Results. arXiv preprint arXiv:2409.10587 (2024)."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW53098.2021.00508"},{"key":"e_1_3_2_1_8_1","volume-title":"Proceedings of the International Conference on Computational Linguistics. 10183-10213","author":"Fan Zhihao","year":"2025","unstructured":"Zhihao Fan, Lai Wei, Jialong Tang, Wei Chen, Wang Siyuan, Zhongyu Wei, and Fei Huang. 2025. AI Hospital: Benchmarking Large Language Models in a Multi-agent Medical Interaction Simulator. In Proceedings of the International Conference on Computational Linguistics. 10183-10213."},{"key":"e_1_3_2_1_9_1","volume-title":"MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models. arXiv preprint arXiv:2306.13394","author":"Fu Chaoyou","year":"2023","unstructured":"Chaoyou Fu, Peixian Chen, Yunhang Shen, Yulei Qin, Mengdan Zhang, Xu Lin, Jinrui Yang, Xiawu Zheng, Ke Li, Xing Sun, et al., 2023. MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models. arXiv preprint arXiv:2306.13394 (2023)."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02245"},{"key":"e_1_3_2_1_11_1","unstructured":"Adam Geitgey. [n.d.]. Face Recognition. https:\/\/github.com\/ageitgey\/face_recognition?tab=readme-ov-file"},{"key":"e_1_3_2_1_12_1","volume-title":"SciAgents: Automating Scientific Discovery Through Bioinspired Multi-Agent Intelligent Graph Reasoning. Advanced Materials","author":"Ghafarollahi Alireza","year":"2024","unstructured":"Alireza Ghafarollahi and Markus J Buehler. 2024. SciAgents: Automating Scientific Discovery Through Bioinspired Multi-Agent Intelligent Graph Reasoning. Advanced Materials (2024), 2413523."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2018.00223"},{"key":"e_1_3_2_1_14_1","unstructured":"Google. 2025. Gemini 2.0 Flash. https:\/\/developers.googleblog.com\/en\/experiment-with-gemini-20-flash-native-image-generation\/"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053928"},{"key":"e_1_3_2_1_16_1","unstructured":"Daya Guo Dejian Yang Haowei Zhang Junxiao Song Ruoyu Zhang Runxin Xu Qihao Zhu Shirong Ma Peiyi Wang Xiao Bi et al. 2025. Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning. arXiv preprint arXiv:2501.12948 (2025)."},{"key":"e_1_3_2_1_17_1","volume-title":"Proceedings of the International Joint Conference on Artificial Intelligence. 8048-8057","author":"Guo Taicheng","year":"2024","unstructured":"Taicheng Guo, Xiuying Chen, Yaqi Wang, Ruidi Chang, Shichao Pei, Nitesh V Chawla, Olaf Wiest, and Xiangliang Zhang. 2024a. Large language model based multi-agents: a survey of progress and challenges. In Proceedings of the International Joint Conference on Artificial Intelligence. 8048-8057."},{"key":"e_1_3_2_1_18_1","volume-title":"Embodied llm agents learn to cooperate in organized teams. arXiv preprint arXiv:2403.12482","author":"Guo Xudong","year":"2024","unstructured":"Xudong Guo, Kaixuan Huang, Jiale Liu, Wenhui Fan, Natalia V\u00e9lez, Qingyun Wu, Huazheng Wang, Thomas L Griffiths, and Mengdi Wang. 2024b. Embodied llm agents learn to cooperate in organized teams. arXiv preprint arXiv:2403.12482 (2024)."},{"key":"e_1_3_2_1_19_1","volume-title":"Mmworld: Towards multi-discipline multi-faceted world model evaluation in videos. arXiv preprint arXiv:2406.08407","author":"He Xuehai","year":"2024","unstructured":"Xuehai He, Weixi Feng, Kaizhi Zheng, Yujie Lu, Wanrong Zhu, Jiachen Li, Yue Fan, Jianfeng Wang, Linjie Li, Zhengyuan Yang, et al., 2024. Mmworld: Towards multi-discipline multi-faceted world model evaluation in videos. arXiv preprint arXiv:2406.08407 (2024)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW59228.2023.00537"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW63382.2024.00332"},{"key":"e_1_3_2_1_22_1","volume-title":"Proceedings of the International Conference on Learning Representations.","author":"Hong Sirui","year":"2024","unstructured":"Sirui Hong, Mingchen Zhuge, Jonathan Chen, Xiawu Zheng, Yuheng Cheng, Jinlin Wang, Ceyao Zhang, Zili Wang, Steven Ka Shing Yau, Zijuan Lin, et al., 2024. MetaGPT: Meta Programming for A Multi-Agent Collaborative Framework. In Proceedings of the International Conference on Learning Representations."},{"key":"e_1_3_2_1_23_1","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition Workshops. 3235-3244","author":"Koshkina Maria","unstructured":"Maria Koshkina and James H. Elder. 2024. A General Framework for Jersey Number Recognition in Sports Video. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition Workshops. 3235-3244."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01263"},{"key":"e_1_3_2_1_25_1","volume-title":"LLaVA-OneVision: Easy Visual Task Transfer. Transactions on Machine Learning Research","author":"Li Bo","year":"2025","unstructured":"Bo Li, Yuanhan Zhang, Dong Guo, Renrui Zhang, Feng Li, Hao Zhang, Kaichen Zhang, Peiyuan Zhang, Yanwei Li, Ziwei Liu, and Chunyuan Li. 2025. LLaVA-OneVision: Easy Visual Task Transfer. Transactions on Machine Learning Research (2025)."},{"key":"e_1_3_2_1_26_1","volume-title":"Hani Itani, Dmitrii Khizbullin, and Bernard Ghanem.","author":"Li Guohao","year":"2023","unstructured":"Guohao Li, Hasan Abed Al Kader Hammoud, Hani Itani, Dmitrii Khizbullin, and Bernard Ghanem. 2023. CAMEL: Communicative Agents for ''Mind'' Exploration of Large Language Model Society. In Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_1_27_1","volume-title":"Sports-qa: A large-scale video question answering benchmark for complex and professional sports. arXiv preprint arXiv:2401.01505","author":"Li Haopeng","year":"2024","unstructured":"Haopeng Li, Andong Deng, Qiuhong Ke, Jun Liu, Hossein Rahmani, Yulan Guo, Bernt Schiele, and Chen Chen. 2024a. Sports-qa: A large-scale video question answering benchmark for complex and professional sports. arXiv preprint arXiv:2401.01505 (2024)."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1007\/s44336-024-00009-2"},{"key":"e_1_3_2_1_29_1","volume-title":"Videochat-flash: Hierarchical compression for long-context video modeling. arXiv preprint arXiv:2501.00574","author":"Li Xinhao","year":"2024","unstructured":"Xinhao Li, Yi Wang, Jiashuo Yu, Xiangyu Zeng, Yuhan Zhu, Haian Huang, Jianfei Gao, Kunchang Li, Yinan He, Chenting Wang, et al., 2024c. Videochat-flash: Hierarchical compression for long-context video modeling. arXiv preprint arXiv:2501.00574 (2024)."},{"key":"e_1_3_2_1_30_1","unstructured":"Aixin Liu Bei Feng Bing Xue Bingxuan Wang Bochao Wu Chengda Lu Chenggang Zhao Chengqi Deng Chenyu Zhang Chong Ruan et al. 2024b. Deepseek-v3 technical report. arXiv preprint arXiv:2412.19437 (2024)."},{"key":"e_1_3_2_1_31_1","unstructured":"Haotian Liu Chunyuan Li Qingyang Wu and Yong Jae Lee. 2023. Visual Instruction Tuning. In Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_1_32_1","volume-title":"Proceedings of the European Conference on Computer Vision.","author":"Liu Shilong","year":"2024","unstructured":"Shilong Liu, Zhaoyang Zeng, Tianhe Ren, Feng Li, Hao Zhang, Jie Yang, Chunyuan Li, Jianwei Yang, Hang Su, Jun Zhu, et al., 2024c. Grounding dino: Marrying dino with grounded pre-training for open-set object detection. In Proceedings of the European Conference on Computer Vision."},{"key":"e_1_3_2_1_33_1","volume-title":"Proceedings of the European Conference on Computer Vision. 216-233","author":"Liu Yuan","year":"2024","unstructured":"Yuan Liu, Haodong Duan, Yuanhan Zhang, Bo Li, Songyang Zhang, Wangbo Zhao, Yike Yuan, Jiaqi Wang, Conghui He, Ziwei Liu, et al., 2024a. Mmbench: Is your multi-modal model an all-around player?. In Proceedings of the European Conference on Computer Vision. 216-233."},{"key":"e_1_3_2_1_34_1","volume-title":"Proceedings of the International Conference on Learning Representations.","author":"Meng Fanqing","year":"2025","unstructured":"Fanqing Meng, Chuanhao Li, Jin Wang, Quanfeng Lu, Hao Tian, Tianshuo Yang, Jiaqi Liao, Xizhou Zhu, Jifeng Dai, Yu Qiao, et al., 2025. MMIU: Multimodal Multi-image Understanding for Evaluating Large Vision-Language Models. In Proceedings of the International Conference on Learning Representations."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW59228.2023.00536"},{"key":"e_1_3_2_1_36_1","unstructured":"OpenAI. 2024. GPT-4o. https:\/\/openai.com"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3583780.3615120"},{"key":"e_1_3_2_1_38_1","first-page":"15174","author":"Qian Chen","year":"2024","unstructured":"Chen Qian, Wei Liu, Hongzhang Liu, Nuo Chen, Yufan Dang, Jiahao Li, Cheng Yang, Weize Chen, Yusheng Su, Xin Cong, et al., 2024. ChatDev: Communicative Agents for Software Development. In Association for Computational Linguistics. 15174-15186.","journal-title":"ChatDev: Communicative Agents for Software Development. In Association for Computational Linguistics."},{"key":"e_1_3_2_1_39_1","volume-title":"Proceedings of the International Conference on Machine Learning.","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al., 2021. Learning transferable visual models from natural language supervision. In Proceedings of the International Conference on Machine Learning."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00785"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.99"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00269"},{"key":"e_1_3_2_1_43_1","first-page":"8634","article-title":"Reflexion: Language agents with verbal reinforcement learning","volume":"36","author":"Shinn Noah","year":"2023","unstructured":"Noah Shinn, Federico Cassano, Ashwin Gopinath, Karthik Narasimhan, and Shunyu Yao. 2023. Reflexion: Language agents with verbal reinforcement learning. In Advances in Neural Information Processing Systems, Vol. 36. 8634-8652.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW63382.2024.00334"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58601-0_39"},{"key":"e_1_3_2_1_46_1","unstructured":"Gemini Team Rohan Anil Sebastian Borgeaud Jean-Baptiste Alayrac Jiahui Yu Radu Soricut Johan Schalkwyk Andrew M Dai Anja Hauth Katie Millican et al. 2023. Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:2312.11805 (2023)."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2017.04.011"},{"key":"e_1_3_2_1_48_1","volume-title":"The Thirty-eight Conference on Neural Information Processing Systems Datasets and Benchmarks Track.","author":"Wang Jize","year":"2024","unstructured":"Jize Wang, Ma Zerun, Yining Li, Songyang Zhang, Cailian Chen, Kai Chen, and Xinyi Le. 2024b. GTA: a benchmark for general tool agents. In The Thirty-eight Conference on Neural Information Processing Systems Datasets and Benchmarks Track."},{"key":"e_1_3_2_1_49_1","first-page":"1","article-title":"TacticAI: an AI assistant for football tactics","volume":"15","author":"Wang Zhe","year":"2024","unstructured":"Zhe Wang, Petar Veli\u010dkovi\u0107, Daniel Hennes, Nenad Toma\u0161ev, Laurel Prince, Michael Kaisers, Yoram Bachrach, Romuald Elie, Li Kevin Wenliang, Federico Piccinini, et al., 2024a. TacticAI: an AI assistant for football tactics. Nature Communications, Vol. 15, 1 (2024), 1-13.","journal-title":"Nature Communications"},{"key":"e_1_3_2_1_50_1","volume-title":"Generative Multi-Agent Collaboration in Embodied AI: A Systematic Review. arXiv preprint arXiv:2502.11518","author":"Wu Di","year":"2025","unstructured":"Di Wu, Xian Wei, Guang Chen, Hao Shen, Xiangfeng Wang, Wenhao Li, and Bo Jin. 2025b. Generative Multi-Agent Collaboration in Embodied AI: A Systematic Review. arXiv preprint arXiv:2502.11518 (2025)."},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19836-6_2"},{"key":"e_1_3_2_1_52_1","volume-title":"SpatialScore: Towards Unified Evaluation for Multimodal Spatial Understanding. arXiv preprint arXiv:2505.17012","author":"Wu Haoning","year":"2025","unstructured":"Haoning Wu, Xiao Huang, Yaohui Chen, Ya Zhang, Yanfeng Wang, and Weidi Xie. 2025a. SpatialScore: Towards Unified Evaluation for Multimodal Spatial Understanding. arXiv preprint arXiv:2505.17012 (2025)."},{"key":"e_1_3_2_1_53_1","volume-title":"First Conference on Language Modeling.","author":"Wu Qingyun","year":"2023","unstructured":"Qingyun Wu, Gagan Bansal, Jieyu Zhang, Yiran Wu, Beibin Li, Erkang Zhu, Li Jiang, Xiaoyun Zhang, Shaokun Zhang, Jiale Liu, et al., 2023. Autogen: Enabling next-gen llm applications via multi-agent conversation. In First Conference on Language Modeling."},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.naacl-long.283"},{"key":"e_1_3_2_1_55_1","volume-title":"Proceedings of the International Conference on Learning Representations.","author":"Xia Haotian","year":"2025","unstructured":"Haotian Xia, Zhengbang Yang, Junbo Zou, Rhys Tracy, Yuqing Wang, Chi Lu, Christopher Lai, Yanjun He, Xun Shao, Zhuoqing Xie, et al., 2025. SPORTU: A Comprehensive Sports Understanding Benchmark for Multimodal Large Language Models. In Proceedings of the International Conference on Learning Representations."},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00296"},{"key":"e_1_3_2_1_57_1","volume-title":"SGA-INTERACT: A 3D Skeleton-based Benchmark for Group Activity Understanding in Modern Basketball Tactic. arXiv preprint arXiv:2503.06522","author":"Yang Yuchen","year":"2025","unstructured":"Yuchen Yang, Wei Wang, Yifei Liu, Linfeng Dong, Hao Wu, Mingxin Zhang, Zhihang Zhong, and Xiao Sun. 2025. SGA-INTERACT: A 3D Skeleton-based Benchmark for Group Activity Understanding in Modern Basketball Tactic. arXiv preprint arXiv:2503.06522 (2025)."},{"key":"e_1_3_2_1_58_1","volume-title":"Proceedings of the International Conference on Learning Representations.","author":"Yao Shunyu","year":"2023","unstructured":"Shunyu Yao, Jeffrey Zhao, Dian Yu, Nan Du, Izhak Shafran, Karthik Narasimhan, and Yuan Cao. 2023. React: Synergizing reasoning and acting in language models. In Proceedings of the International Conference on Learning Representations."},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00913"},{"key":"e_1_3_2_1_60_1","volume-title":"Mmmu-pro: A more robust multi-discipline multimodal understanding benchmark. arXiv preprint arXiv:2409.02813","author":"Yue Xiang","year":"2024","unstructured":"Xiang Yue, Tianyu Zheng, Yuansheng Ni, Yubo Wang, Kai Zhang, Shengbang Tong, Yuxuan Sun, Botao Yu, Ge Zhang, Huan Sun, et al., 2024b. Mmmu-pro: A more robust multi-discipline multimodal understanding benchmark. arXiv preprint arXiv:2409.02813 (2024)."},{"key":"e_1_3_2_1_61_1","unstructured":"Boqiang Zhang Kehan Li Zesen Cheng Zhiqiang Hu Yuqian Yuan Guanzheng Chen Sicong Leng Yuming Jiang Hang Zhang Xin Li et al. 2025. VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding. arXiv preprint arXiv:2501.13106 (2025)."},{"key":"e_1_3_2_1_62_1","volume-title":"Video instruction tuning with synthetic data. arXiv preprint arXiv:2410.02713","author":"Zhang Yuanhan","year":"2024","unstructured":"Yuanhan Zhang, Jinming Wu, Wei Li, Bo Li, Zejun Ma, Ziwei Liu, and Chunyuan Li. 2024. Video instruction tuning with synthetic data. arXiv preprint arXiv:2410.02713 (2024)."},{"key":"e_1_3_2_1_63_1","volume-title":"Mlvu: A comprehensive benchmark for multi-task long video understanding. arXiv preprint arXiv:2406.04264","author":"Zhou Junjie","year":"2024","unstructured":"Junjie Zhou, Yan Shu, Bo Zhao, Boya Wu, Shitao Xiao, Xi Yang, Yongping Xiong, Bo Zhang, Tiejun Huang, and Zheng Liu. 2024. Mlvu: A comprehensive benchmark for multi-task long video understanding. arXiv preprint arXiv:2406.04264 (2024)."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755144","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:55:13Z","timestamp":1765310113000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755144"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":63,"alternative-id":["10.1145\/3746027.3755144","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755144","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}