{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,2]],"date-time":"2026-02-02T18:03:30Z","timestamp":1770055410869,"version":"3.49.0"},"publisher-location":"New York, NY, USA","reference-count":40,"publisher":"ACM","funder":[{"name":"Achievements of the 2025 Beijing Municipal Education Science Planning Youth Special Project","award":["CCGA25154"],"award-info":[{"award-number":["CCGA25154"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,11,12]]},"DOI":"10.1145\/3784833.3784850","type":"proceedings-article","created":{"date-parts":[[2026,2,2]],"date-time":"2026-02-02T05:22:31Z","timestamp":1770009751000},"page":"114-122","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["BasketVision: Benchmarking MLLMs' Grasp of Complex Dynamic Systems"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-4061-1241","authenticated-orcid":false,"given":"Xiankun","family":"Jiang","sequence":"first","affiliation":[{"name":"International School, Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-9191-6320","authenticated-orcid":false,"given":"Qiyao","family":"Sun","sequence":"additional","affiliation":[{"name":"International School, Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-0145-2047","authenticated-orcid":false,"given":"Xiongce","family":"Lv","sequence":"additional","affiliation":[{"name":"Department of Physical Education, Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2026,2]]},"reference":[{"key":"e_1_3_3_1_2_2","doi-asserted-by":"publisher","unstructured":"A. Adel and N. Alani. 2025. Can generative AI reliably synthesise literature? exploring hallucination issues in ChatGPT. AI & Society (2025). 10.1007\/s00146-025-02406-7","DOI":"10.1007\/s00146-025-02406-7"},{"key":"e_1_3_3_1_3_2","unstructured":"Amazon AGI. 2025. The Amazon Nova Family of Models: Technical Report and Model Card. arxiv:https:\/\/arXiv.org\/abs\/2506.12103\u00a0[cs.AI] https:\/\/arxiv.org\/abs\/2506.12103"},{"key":"e_1_3_3_1_4_2","unstructured":"Pravesh Agrawal Szymon Antoniak Emma\u00a0Bou Hanna Baptiste Bout et\u00a0al. 2024. Pixtral 12B. arxiv:https:\/\/arXiv.org\/abs\/2410.07073\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2410.07073"},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.279"},{"key":"e_1_3_3_1_6_2","unstructured":"Shuai Bai Keqin Chen Xuejing Liu Jialin Wang et\u00a0al. 2025. Qwen2.5-VL Technical Report. arxiv:https:\/\/arXiv.org\/abs\/2502.13923\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2502.13923"},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"publisher","DOI":"10.1145\/3442188.3445922"},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"publisher","unstructured":"Mika\u00ebl Chelli Jules Descamps Vincent Lavou\u00e9 Christophe Trojani et\u00a0al. 2024. Hallucination Rates and Reference Accuracy of ChatGPT and Bard for Systematic Reviews: Comparative Analysis. J Med Internet Res 26 (22 May 2024) e53164. 10.2196\/53164","DOI":"10.2196\/53164"},{"key":"e_1_3_3_1_9_2","doi-asserted-by":"publisher","unstructured":"Wey\u00a0Yeh Choong Yangyang Guo and Mohan\u00a0S. Kankanhalli. 2024. VidHal: Benchmarking Temporal Hallucinations in Vision LLMs. CoRR abs\/2411.16771 (2024). arXiv:https:\/\/arXiv.org\/abs\/2411.1677110.48550\/ARXIV.2411.16771","DOI":"10.48550\/ARXIV.2411.16771"},{"key":"e_1_3_3_1_10_2","volume-title":"9th International Conference on Learning Representations, ICLR 2021, Virtual Event, Austria, May 3-7, 2021","author":"Dosovitskiy Alexey","year":"2021","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, et\u00a0al. 2021. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. In 9th International Conference on Learning Representations, ICLR 2021, Virtual Event, Austria, May 3-7, 2021. OpenReview.net. https:\/\/openreview.net\/forum?id=YicbFdNTTy"},{"key":"e_1_3_3_1_11_2","volume-title":"Advances in Neural Information Processing Systems 38: Annual Conference on Neural Information Processing Systems 2024, NeurIPS 2024, Vancouver, BC, Canada, December 10 - 15, 2024","author":"Fang Xinyu","year":"2024","unstructured":"Xinyu Fang, Kangrui Mao, Haodong Duan, Xiangyu Zhao, et\u00a0al. 2024. MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding. In Advances in Neural Information Processing Systems 38: Annual Conference on Neural Information Processing Systems 2024, NeurIPS 2024, Vancouver, BC, Canada, December 10 - 15, 2024, Amir Globersons, Lester Mackey, Danielle Belgrave, Angela Fan, Ulrich Paquet, Jakub\u00a0M. Tomczak, and Cheng Zhang (Eds.). http:\/\/papers.nips.cc\/paper_files\/paper\/2024\/hash\/a2326c9715a516c91174132e0170073a-Abstract-Datasets_and_Benchmarks_Track.html"},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"publisher","unstructured":"Chaoyou Fu Peixian Chen Yunhang Shen Yulei Qin et\u00a0al. 2023. MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models. CoRR abs\/2306.13394 (2023). arXiv:https:\/\/arXiv.org\/abs\/2306.1339410.48550\/ARXIV.2306.13394","DOI":"10.48550\/ARXIV.2306.13394"},{"key":"e_1_3_3_1_13_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02245"},{"key":"e_1_3_3_1_14_2","volume-title":"7th International Conference on Learning Representations, ICLR 2019, New Orleans, LA, USA, May 6-9, 2019","author":"Gao Jun","year":"2019","unstructured":"Jun Gao, Di He, Xu Tan, Tao Qin, et\u00a0al. 2019. Representation Degeneration Problem in Training Natural Language Generation Models. In 7th International Conference on Learning Representations, ICLR 2019, New Orleans, LA, USA, May 6-9, 2019. OpenReview.net. https:\/\/openreview.net\/forum?id=SkEYojRqtm"},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.670"},{"key":"e_1_3_3_1_16_2","unstructured":"Jared Kaplan Sam McCandlish Tom Henighan Tom\u00a0B. Brown et\u00a0al. 2020. Scaling Laws for Neural Language Models. CoRR abs\/2001.08361 (2020). arXiv:https:\/\/arXiv.org\/abs\/2001.08361https:\/\/arxiv.org\/abs\/2001.08361"},{"key":"e_1_3_3_1_17_2","doi-asserted-by":"publisher","unstructured":"Dingming Li Hongxing Li Zixuan Wang Yuchen Yan et\u00a0al. 2025. ViewSpatial-Bench: Evaluating Multi-perspective Spatial Localization in Vision-Language Models. CoRR abs\/2505.21500 (2025). arXiv:https:\/\/arXiv.org\/abs\/2505.2150010.48550\/ARXIV.2505.21500","DOI":"10.48550\/ARXIV.2505.21500"},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"publisher","unstructured":"Jian Li and Weiheng Lu. 2024. A Survey on Benchmarks of Multimodal Large Language Models. CoRR abs\/2408.08632 (2024). arXiv:https:\/\/arXiv.org\/abs\/2408.0863210.48550\/ARXIV.2408.08632","DOI":"10.48550\/ARXIV.2408.08632"},{"key":"e_1_3_3_1_19_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02095"},{"key":"e_1_3_3_1_20_2","doi-asserted-by":"publisher","unstructured":"Jingli Lin Chenming Zhu Runsen Xu Xiaohan Mao et\u00a0al. 2025. OST-Bench: Evaluating the Capabilities of MLLMs in Online Spatio-temporal Scene Understanding. CoRR abs\/2507.07984 (2025). arXiv:https:\/\/arXiv.org\/abs\/2507.0798410.48550\/ARXIV.2507.07984","DOI":"10.48550\/ARXIV.2507.07984"},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"publisher","unstructured":"Siqi Liu Guy Lever Zhe Wang Josh Merel et\u00a0al. 2022. From motor control to team play in simulated humanoid football. Sci. Robotics 7 69 (2022). 10.1126\/SCIROBOTICS.ABO0235","DOI":"10.1126\/SCIROBOTICS.ABO0235"},{"key":"e_1_3_3_1_22_2","doi-asserted-by":"publisher","DOI":"10.1109\/CAI59869.2024.00033"},{"key":"e_1_3_3_1_23_2","unstructured":"MiniMax. 2025. MiniMax-01: Scaling Foundation Models with Lightning Attention. arxiv:https:\/\/arXiv.org\/abs\/2501.08313\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2501.08313"},{"key":"e_1_3_3_1_24_2","doi-asserted-by":"publisher","unstructured":"Munan Ning Bin Zhu Yujia Xie Bin Lin et\u00a0al. 2023. Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models. CoRR abs\/2311.16103 (2023). arXiv:https:\/\/arXiv.org\/abs\/2311.1610310.48550\/ARXIV.2311.16103","DOI":"10.48550\/ARXIV.2311.16103"},{"key":"e_1_3_3_1_25_2","unstructured":"OpenAI. 2024. GPT-4o System Card. arxiv:https:\/\/arXiv.org\/abs\/2410.21276\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2410.21276"},{"key":"e_1_3_3_1_26_2","volume-title":"Advances in Neural Information Processing Systems 35: Annual Conference on Neural Information Processing Systems 2022, NeurIPS 2022, New Orleans, LA, USA, November 28 - December 9, 2022","author":"Ouyang Long","year":"2022","unstructured":"Long Ouyang, Jeffrey Wu, Xu Jiang, Diogo Almeida, et\u00a0al. 2022. Training language models to follow instructions with human feedback. In Advances in Neural Information Processing Systems 35: Annual Conference on Neural Information Processing Systems 2022, NeurIPS 2022, New Orleans, LA, USA, November 28 - December 9, 2022, Sanmi Koyejo, S.\u00a0Mohamed, A.\u00a0Agarwal, Danielle Belgrave, K.\u00a0Cho, and A.\u00a0Oh (Eds.). http:\/\/papers.nips.cc\/paper_files\/paper\/2022\/hash\/b1efde53be364a73914f58805a001731-Abstract-Conference.html"},{"key":"e_1_3_3_1_27_2","doi-asserted-by":"publisher","unstructured":"Bryan\u00a0A. Plummer Liwei Wang Chris\u00a0M. Cervantes Juan\u00a0C. Caicedo Julia Hockenmaier and Svetlana Lazebnik. 2017. Flickr30k Entities: Collecting Region-to-Phrase Correspondences for Richer Image-to-Sentence Models. Int. J. Comput. Vis. 123 1 (2017) 74\u201393. 10.1007\/S11263-016-0965-7","DOI":"10.1007\/S11263-016-0965-7"},{"key":"e_1_3_3_1_28_2","unstructured":"Yujia Qin Yining Ye Junjie Fang Haoming Wang et\u00a0al. 2025. UI-TARS: Pioneering Automated GUI Interaction with Native Agents. arxiv:https:\/\/arXiv.org\/abs\/2501.12326\u00a0[cs.AI] https:\/\/arxiv.org\/abs\/2501.12326"},{"key":"e_1_3_3_1_29_2","unstructured":"Gemini Team. 2024. Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context. arxiv:https:\/\/arXiv.org\/abs\/2403.05530\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2403.05530"},{"key":"e_1_3_3_1_30_2","doi-asserted-by":"publisher","DOI":"10.65109\/AHMK4753"},{"key":"e_1_3_3_1_31_2","doi-asserted-by":"publisher","unstructured":"Siting Wang Luoyang Sun Cheng Deng Kun Shao et\u00a0al. 2025. SpatialViz-Bench: Automatically Generated Spatial Visualization Reasoning Tasks for MLLMs. CoRR abs\/2507.07610 (2025). arXiv:https:\/\/arXiv.org\/abs\/2507.0761010.48550\/ARXIV.2507.07610","DOI":"10.48550\/ARXIV.2507.07610"},{"key":"e_1_3_3_1_32_2","doi-asserted-by":"publisher","unstructured":"Yuxuan Wang Yueqian Wang Dongyan Zhao Cihang Xie and Zilong Zheng. 2024. VideoHallucer: Evaluating Intrinsic and Extrinsic Hallucinations in Large Video-Language Models. CoRR abs\/2406.16338 (2024). arXiv:https:\/\/arXiv.org\/abs\/2406.1633810.48550\/ARXIV.2406.16338","DOI":"10.48550\/ARXIV.2406.16338"},{"key":"e_1_3_3_1_33_2","unstructured":"Jason Wei Yi Tay Rishi Bommasani Colin Raffel et\u00a0al. 2022. Emergent Abilities of Large Language Models. Trans. Mach. Learn. Res. 2022 (2022). https:\/\/openreview.net\/forum?id=yzkSU5zdwD"},{"key":"e_1_3_3_1_34_2","doi-asserted-by":"publisher","unstructured":"Lingrui Xu Mandi Liu and Lei Zhang. 2025. TacticExpert: Spatial-Temporal Graph Language Model for Basketball Tactics. CoRR abs\/2503.10722 (2025). arXiv:https:\/\/arXiv.org\/abs\/2503.1072210.48550\/ARXIV.2503.10722","DOI":"10.48550\/ARXIV.2503.10722"},{"key":"e_1_3_3_1_35_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00994"},{"key":"e_1_3_3_1_36_2","doi-asserted-by":"publisher","unstructured":"Shukang Yin Chaoyou Fu Sirui Zhao Ke Li et\u00a0al. 2023. A Survey on Multimodal Large Language Models. CoRR abs\/2306.13549 (2023). arXiv:https:\/\/arXiv.org\/abs\/2306.1354910.48550\/ARXIV.2306.13549","DOI":"10.48550\/ARXIV.2306.13549"},{"key":"e_1_3_3_1_37_2","volume-title":"Advances in Neural Information Processing Systems 33: Annual Conference on Neural Information Processing Systems 2020, NeurIPS 2020, December 6-12, 2020, virtual","author":"Yu Tianhe","year":"2020","unstructured":"Tianhe Yu, Saurabh Kumar, Abhishek Gupta, Sergey Levine, et\u00a0al. 2020. Gradient Surgery for Multi-Task Learning. In Advances in Neural Information Processing Systems 33: Annual Conference on Neural Information Processing Systems 2020, NeurIPS 2020, December 6-12, 2020, virtual, Hugo Larochelle, Marc\u2019Aurelio Ranzato, Raia Hadsell, Maria-Florina Balcan, and Hsuan-Tien Lin (Eds.). https:\/\/proceedings.neurips.cc\/paper\/2020\/hash\/3fe78a8acf5fda99de95303940a2420c-Abstract.html"},{"key":"e_1_3_3_1_38_2","volume-title":"Forty-first International Conference on Machine Learning, ICML 2024, Vienna, Austria, July 21-27, 2024","author":"Yu Weihao","year":"2024","unstructured":"Weihao Yu, Zhengyuan Yang, Linjie Li, Jianfeng Wang, et\u00a0al. 2024. MM-Vet: Evaluating Large Multimodal Models for Integrated Capabilities. In Forty-first International Conference on Machine Learning, ICML 2024, Vienna, Austria, July 21-27, 2024. OpenReview.net. https:\/\/openreview.net\/forum?id=KOTutrSR2y"},{"key":"e_1_3_3_1_39_2","doi-asserted-by":"publisher","unstructured":"Jiacheng Zhang Yang Jiao Shaoxiang Chen Jingjing Chen and Yu-Gang Jiang. 2024. EventHallusion: Diagnosing Event Hallucinations in Video LLMs. CoRR abs\/2409.16597 (2024). arXiv:https:\/\/arXiv.org\/abs\/2409.1659710.48550\/ARXIV.2409.16597","DOI":"10.48550\/ARXIV.2409.16597"},{"key":"e_1_3_3_1_40_2","doi-asserted-by":"publisher","unstructured":"Wanyue Zhang Yibin Huang Yangbin Xu JingJing Huang et\u00a0al. 2025. Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture. CoRR abs\/2509.02359 (2025). arXiv:https:\/\/arXiv.org\/abs\/2509.0235910.48550\/ARXIV.2509.02359","DOI":"10.48550\/ARXIV.2509.02359"},{"key":"e_1_3_3_1_41_2","doi-asserted-by":"publisher","unstructured":"Shijie Zhou Alexander Vilesov Xuehai He Ziyu Wan et\u00a0al. 2025. VLM4D: Towards Spatiotemporal Awareness in Vision Language Models. CoRR abs\/2508.02095 (2025). arXiv:https:\/\/arXiv.org\/abs\/2508.0209510.48550\/ARXIV.2508.02095","DOI":"10.48550\/ARXIV.2508.02095"}],"event":{"name":"ICCIP 2025: 2025 the 11th International Conference on Communication and Information Processing","location":"Lingshui Hainan China","acronym":"ICCIP 2025"},"container-title":["Proceedings of the 2025 11th International Conference on Communication and Information Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3784833.3784850","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,2]],"date-time":"2026-02-02T07:46:09Z","timestamp":1770018369000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3784833.3784850"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,12]]},"references-count":40,"alternative-id":["10.1145\/3784833.3784850","10.1145\/3784833"],"URL":"https:\/\/doi.org\/10.1145\/3784833.3784850","relation":{},"subject":[],"published":{"date-parts":[[2025,11,12]]},"assertion":[{"value":"2026-02-01","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}