{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:16:08Z","timestamp":1765307768582,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":47,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62176016,72274127"],"award-info":[{"award-number":["62176016,72274127"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Haidian innovation and translation program from Peking University Third Hospital","award":["HDCXZHKC2023203"],"award-info":[{"award-number":["HDCXZHKC2023203"]}]},{"name":"Capital Health Development Research Project","award":["2022-2-2013"],"award-info":[{"award-number":["2022-2-2013"]}]},{"name":"Research on the Decision Support System for Urban and Park Carbon Emissions Empowered by Digital Technology"},{"name":"Guizhou Province Science and Technology Project","award":["Qiankehe[2024] General 058"],"award-info":[{"award-number":["Qiankehe[2024] General 058"]}]},{"name":"National Key R\\&D Program of China","award":["2021YFB2104800"],"award-info":[{"award-number":["2021YFB2104800"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755556","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T05:44:48Z","timestamp":1761371088000},"page":"4698-4707","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["EventFormer: A Node-graph Hierarchical Attention Transformer for Action-centric Video Event Prediction"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7761-3636","authenticated-orcid":false,"given":"Qile","family":"Su","sequence":"first","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-6800-0908","authenticated-orcid":false,"given":"Shoutai","family":"Zhu","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-8549-3179","authenticated-orcid":false,"given":"Shuai","family":"Zhang","sequence":"additional","affiliation":[{"name":"University of Science and Technology Beijing, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2365-589X","authenticated-orcid":false,"given":"Baoyu","family":"Liang","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4414-4965","authenticated-orcid":false,"given":"Chao","family":"Tong","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, Beihang University, Beijing, China and State Key Laboratory of Virtual Reality Technology and Systems, Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Shuai Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge Sibo Song Kai Dang Peng Wang Shijie Wang Jun Tang Humen Zhong Yuanzhi Zhu Mingkun Yang Zhaohai Li JianqiangWan PengfeiWang Wei Ding Zheren Fu Yiheng Xu Jiabo Ye Xi Zhang Tianbao Xie Zesen Cheng Hang Zhang Zhibo Yang Haiyang Xu and Junyang Lin. 2025. Qwen2.5-VL Technical Report. arXiv:2502.13923 [cs.CV] https:\/\/arxiv.org\/abs\/2502.13923"},{"key":"e_1_3_2_1_2_1","volume-title":"Proceedings of ACL-08: HLT. 789--797","author":"Chambers Nathanael","year":"2008","unstructured":"Nathanael Chambers and Dan Jurafsky. 2008. Unsupervised learning of narrative event chains. In Proceedings of ACL-08: HLT. 789--797."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.3115\/1690219.1690231"},{"key":"e_1_3_2_1_4_1","volume-title":"Tempura: Temporal event masked prediction and understanding for reasoning in action. arXiv preprint arXiv:2505.01583","author":"Cheng Jen-Hao","year":"2025","unstructured":"Jen-Hao Cheng, Vivian Wang, HuayuWang, Huapeng Zhou, Yi-Hao Peng, Hou-I Liu, Hsiang-Wei Huang, Kuang-Ming Chen, Cheng-Yen Yang, Wenhao Chai, et al. 2025. Tempura: Temporal event masked prediction and understanding for reasoning in action. arXiv preprint arXiv:2505.01583 (2025)."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2020.2991965"},{"key":"e_1_3_2_1_6_1","volume-title":"International conference on machine learning. PMLR, 1243--1252","author":"Gehring Jonas","year":"2017","unstructured":"Jonas Gehring, Michael Auli, David Grangier, Denis Yarats, and Yann N Dauphin. 2017. Convolutional sequence to sequence learning. In International conference on machine learning. PMLR, 1243--1252."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01325"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2022.3213652"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v30i1.10344"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01842"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_12_1","volume-title":"Devendra Singh Dhami, and Kristian Kersting","author":"Helff Lukas","year":"2024","unstructured":"Lukas Helff, Wolfgang Stammer, Hikaru Shindo, Devendra Singh Dhami, and Kristian Kersting. 2024. V-LoL: A Diagnostic Dataset for Visual Logical Learning. arXiv:2306.07743 [cs.AI] https:\/\/arxiv.org\/abs\/2306.07743"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.5555\/2380816.2380858"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01025"},{"key":"e_1_3_2_1_15_1","volume-title":"European Conference on Computer Vision. Springer, 140--158","author":"Kim Sanghwan","year":"2024","unstructured":"Sanghwan Kim, Daoji Huang, Yongqin Xian, Otmar Hilliges, Luc Van Gool, and Xi Wang. 2024. Palm: Predicting actions through language models. In European Conference on Computer Vision. Springer, 140--158."},{"key":"e_1_3_2_1_16_1","volume-title":"Semi-supervised classification with graph convolutional networks. arXiv preprint arXiv:1609.02907","author":"Kipf Thomas N","year":"2016","unstructured":"Thomas N Kipf and MaxWelling. 2016. Semi-supervised classification with graph convolutional networks. arXiv preprint arXiv:1609.02907 (2016)."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-022-01594-9"},{"key":"e_1_3_2_1_18_1","first-page":"1","article-title":"A Higher order Markov model for time series forecasting","volume":"57","author":"Ky Dao Xuan","year":"2018","unstructured":"Dao Xuan Ky and Luc Tri Tuyen. 2018. A Higher order Markov model for time series forecasting. International Journal of Applied Mathematics and Statistics 57, 3 (2018), 1--18.","journal-title":"International Journal of Applied Mathematics and Statistics"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.706"},{"key":"e_1_3_2_1_20_1","unstructured":"Baoyu Liang Qile Su Shoutai Zhu Yuchen Liang and Chao Tong. 2025. VidEvent: A Large Dataset for Understanding Dynamic Evolution of Events in Videos. arXiv:2506.02448 [cs.CV] https:\/\/arxiv.org\/abs\/2506.02448"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.342"},{"key":"e_1_3_2_1_22_1","volume-title":"Proceedings, Part I 16","author":"Liu Miao","year":"2020","unstructured":"Miao Liu, Siyu Tang, Yin Li, and James M Rehg. 2020. Forecasting human-object interaction: joint prediction of motor attention and actions in first person video. In Computer Vision--ECCV 2020: 16th European Conference, Glasgow, UK, August 23--28, 2020, Proceedings, Part I 16. Springer, 704--721."},{"key":"e_1_3_2_1_23_1","volume-title":"European Conference on Computer Vision. Springer, 38--55","author":"Liu Shilong","year":"2024","unstructured":"Shilong Liu, Zhaoyang Zeng, Tianhe Ren, Feng Li, Hao Zhang, Jie Yang, Qing Jiang, Chunyuan Li, Jianwei Yang, Hang Su, et al. 2024. Grounding dino: Marrying dino with grounded pre-training for open-set object detection. In European Conference on Computer Vision. Springer, 38--55."},{"key":"e_1_3_2_1_24_1","unstructured":"Raymond J Mooney and Gerald DeJong. 1985. Learning schemata for natural language processing.. In Ijcai. 681--687."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v31i1.11237"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/IRC.2020.00085"},{"key":"e_1_3_2_1_27_1","volume-title":"Proceedings of the 14th Conference of the European Chapter of the Association for Computational Linguistics. 220--229","author":"Pichotta Karl","year":"2014","unstructured":"Karl Pichotta and Raymond Mooney. 2014. Statistical script learning with multiargument events. In Proceedings of the 14th Conference of the European Chapter of the Association for Computational Linguistics. 220--229."},{"key":"e_1_3_2_1_28_1","volume-title":"STEP: Enhancing Video-LLMs' Compositional Reasoning by Spatio-Temporal Graph-guided Self-Training. arXiv:2412.00161 [cs.CV] https:\/\/arxiv.org\/abs\/2412.00161","author":"Qiu Haiyi","year":"2025","unstructured":"Haiyi Qiu, Minghe Gao, Long Qian, Kaihang Pan, Qifan Yu, Juncheng Li, Wenjie Wang, Siliang Tang, Yueting Zhuang, and Tat-Seng Chua. 2025. STEP: Enhancing Video-LLMs' Compositional Reasoning by Spatio-Temporal Graph-guided Self-Training. arXiv:2412.00161 [cs.CV] https:\/\/arxiv.org\/abs\/2412.00161"},{"key":"e_1_3_2_1_29_1","volume-title":"International conference on machine learning. PmLR, 8748--8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al. 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PmLR, 8748--8763."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01762"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00660"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/S15-1024"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1140\/epjds\/s13688-018-0171-7"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00554"},{"key":"e_1_3_2_1_35_1","volume-title":"Proceedings of the 4th International Joint Conference on Artificial Intelligence. IJCAI","volume":"1","author":"Schank Roger C","year":"1975","unstructured":"Roger C Schank and Robert P Abelson. 1975. Scripts, plans and goals. In Proceedings of the 4th International Joint Conference on Artificial Intelligence. IJCAI, Vol. 1."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2023.127063"},{"key":"e_1_3_2_1_37_1","volume-title":"Attention is all you need. Advances in neural information processing systems 30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_1_38_1","volume-title":"GRAPH ATTENTION NETWORKS. stat 1050","author":"Velickovic Petar","year":"2018","unstructured":"Petar Velickovic, Guillem Cucurull, Arantxa Casanova, Adriana Romero, Pietro Lio, and Yoshua Bengio. 2018. GRAPH ATTENTION NETWORKS. stat 1050 (2018), 4."},{"key":"e_1_3_2_1_39_1","unstructured":"Haonan Wang Hongfu Liu Xiangyan Liu Chao Du Kenji Kawaguchi Ye Wang and Tianyu Pang. 2025. Fostering Video Reasoning via Next-Event Prediction. arXiv:2505.22457 [cs.CV] https:\/\/arxiv.org\/abs\/2505.22457"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.crac-1.14"},{"key":"e_1_3_2_1_41_1","volume-title":"Highly-Performant Package for Graph Neural Networks. arXiv preprint arXiv:1909.01315","author":"Wang Minjie","year":"2019","unstructured":"Minjie Wang, Da Zheng, Zihao Ye, Quan Gan, Mufei Li, Xiang Song, Jinjing Zhou, Chao Ma, Lingfan Yu, Yu Gai, Tianjun Xiao, Tong He, George Karypis, Jinyang Li, and Zheng Zhang. 2019. Deep Graph Library: A Graph-Centric, Highly-Performant Package for Graph Neural Networks. arXiv preprint arXiv:1909.01315 (2019)."},{"key":"e_1_3_2_1_42_1","volume-title":"Caption anything: Interactive image description with diverse multimodal controls. arXiv preprint arXiv:2305.02677","author":"Wang Teng","year":"2023","unstructured":"Teng Wang, Jinrui Zhang, Junjie Fei, Hao Zheng, Yunlong Tang, Zhe Li, Mingqi Gao, and Shanshan Zhao. 2023. Caption anything: Interactive image description with diverse multimodal controls. arXiv preprint arXiv:2305.02677 (2023)."},{"key":"e_1_3_2_1_43_1","volume-title":"How powerful are graph neural networks? arXiv preprint arXiv:1810.00826","author":"Xu Keyulu","year":"2018","unstructured":"Keyulu Xu, Weihua Hu, Jure Leskovec, and Stefanie Jegelka. 2018. How powerful are graph neural networks? arXiv preprint arXiv:1810.00826 (2018)."},{"volume-title":"Big Data & Cloud Computing, Sustainable Computing & Communications, Social Computing & Networking (ISPA\/BDCloud\/SocialCom\/SustainCom)","author":"Yang Shuang","key":"e_1_3_2_1_44_1","unstructured":"Shuang Yang, FaliWang, Daren Zha, Cong Xue, and Zhihao Tang. 2020. NarGNN: Narrative graph neural networks for new script event prediction problem. In 2020 IEEE Intl Conf on Parallel & Distributed Processing with Applications, Big Data & Cloud Computing, Sustainable Computing & Communications, Social Computing & Networking (ISPA\/BDCloud\/SocialCom\/SustainCom). IEEE, 481--488."},{"key":"e_1_3_2_1_45_1","volume-title":"Wildes","author":"Zhao He","year":"2021","unstructured":"He Zhao and Richard P. Wildes. 2021. Review of Video Predictive Understanding: Early Action Recognition and Future Action Prediction. arXiv:2107.05140 [cs.CV] https:\/\/arxiv.org\/abs\/2107.05140"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/3444689"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.coling-main.29"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755556","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:13:39Z","timestamp":1765307619000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755556"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":47,"alternative-id":["10.1145\/3746027.3755556","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755556","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}