{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:46:55Z","timestamp":1765309615574,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":47,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3763762","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T06:56:44Z","timestamp":1761375404000},"page":"14086-14093","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Higher-Order Vision-Language Fusion for Video Popularity Prediction"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5997-5169","authenticated-orcid":false,"given":"Kele","family":"Xu","sequence":"first","affiliation":[{"name":"College of Computer Science and Technology, National University of Defense Technology, Changsha, Hunan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4141-5950","authenticated-orcid":false,"given":"Qisheng","family":"Xu","sequence":"additional","affiliation":[{"name":"College of Computer Science and Technology, National University of Defense Technology, Changsha, Hunan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-9590-344X","authenticated-orcid":false,"given":"Binli","family":"Luo","sequence":"additional","affiliation":[{"name":"Central South University, Changsha, Hunan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-5808-5695","authenticated-orcid":false,"given":"Han","family":"Zhou","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, Zhejiang, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-9509-0235","authenticated-orcid":false,"given":"Zengming","family":"Lin","sequence":"additional","affiliation":[{"name":"College of Computer Science and Technology, National University of Defense Technology, Changsha, Hunan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-7046-2235","authenticated-orcid":false,"given":"Hui","family":"Geng","sequence":"additional","affiliation":[{"name":"College of Computer Science and Technology, National University of Defense Technology, Changsha, Hunan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-1719-5248","authenticated-orcid":false,"given":"Xianhan","family":"Tan","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, Zhejiang, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2018.2798607"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/2783258.2783348"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2020.3048428"},{"key":"e_1_3_2_1_4_1","volume-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics (NAACL). 4171-4186","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics (NAACL). 4171-4186."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11704-019-8208-z"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-021-11610-8"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01911"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/OJCS.2025.3546754"},{"key":"e_1_3_2_1_9_1","volume-title":"AST: Audio Spectrogram Transformer. Interspeech 2021","author":"Gong Yuan","year":"2021","unstructured":"Yuan Gong, Yu-An Chung, and James Glass. 2021. AST: Audio Spectrogram Transformer. Interspeech 2021 (2021)."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3551593"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2020.2986579"},{"key":"e_1_3_2_1_12_1","volume-title":"Lightgbm: A highly efficient gradient boosting decision tree. Advances in neural information processing systems","author":"Ke Guolin","year":"2017","unstructured":"Guolin Ke, Qi Meng, Thomas Finley, Taifeng Wang, Wei Chen, Weidong Ma, Qiwei Ye, and Tie-Yan Liu. 2017. Lightgbm: A highly efficient gradient boosting decision tree. Advances in neural information processing systems, Vol. 30 (2017)."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/2783258.2788582"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3627704"},{"key":"e_1_3_2_1_15_1","volume-title":"Proceedings of the 39th International Conference on Machine Learning (ICML). 12888-12900","author":"Li Junnan","year":"2022","unstructured":"Junnan Li, Dongxu Li, Caiming Xiong, and Steven CH Hoi. 2022. BLIP: Bootstrapping Language-Image Pre-training for Unified Vision-Language Understanding and Generation. In Proceedings of the 39th International Conference on Machine Learning (ICML). 12888-12900."},{"key":"e_1_3_2_1_16_1","first-page":"700","article-title":"Unsupervised Image-to-Image Translation Networks","author":"Liu Ming-Yu","year":"2017","unstructured":"Ming-Yu Liu, Thomas Breuel, and Jan Kautz. 2017. Unsupervised Image-to-Image Translation Networks. In Advances in Neural Information Processing Systems (NeurIPS). 700-708.","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_2_1_17_1","volume-title":"Proceedings of the 2025 ACM International Conference on Multimedia (MM). 1452-1463","author":"Liu Xinyu","year":"2025","unstructured":"Xinyu Liu, Wei Zhang, Haoyu Chen, and Yong Li. 2025. Multimodal Forecasting of Short-Form Video Engagement: Combining Visual, Textual, and Tabular Features. In Proceedings of the 2025 ACM International Conference on Multimedia (MM). 1452-1463."},{"key":"e_1_3_2_1_18_1","volume-title":"Advances in Neural Information Processing Systems (NeurIPS)","volume":"32","author":"Lu Jiasen","year":"2019","unstructured":"Jiasen Lu, Dhruv Batra, Devi Parikh, and Stefan Lee. 2019. ViLBERT: Pretraining Task-Agnostic Visiolinguistic Representations for Vision-and-Language Tasks. In Advances in Neural Information Processing Systems (NeurIPS), Vol. 32."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612839"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3289600.3291598"},{"key":"e_1_3_2_1_21_1","first-page":"390","volume-title":"Proceedings of the International AAAI Conference on Web and Social Media","volume":"7","author":"Momeni Elaheh","year":"2013","unstructured":"Elaheh Momeni, Claire Cardie, and Myle Ott. 2013. Properties, prediction, and prevalence of useful user-generated comments for descriptive annotation of social media objects. In Proceedings of the International AAAI Conference on Web and Social Media, Vol. 7. 390-399."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3269206.3271781"},{"key":"e_1_3_2_1_23_1","volume-title":"Anna Veronika Dorogush, and Andrey Gulin","author":"Prokhorenkova Liudmila","year":"2018","unstructured":"Liudmila Prokhorenkova, Gleb Gusev, Aleksandr Vorobev, Anna Veronika Dorogush, and Andrey Gulin. 2018. CatBoost: unbiased boosting with categorical features. Advances in neural information processing systems, Vol. 31 (2018)."},{"key":"e_1_3_2_1_24_1","volume-title":"Personalized recommendation combining user interest and social circle","author":"Qian Xueming","year":"2013","unstructured":"Xueming Qian, He Feng, Guoshuai Zhao, and Tao Mei. 2013. Personalized recommendation combining user interest and social circle. IEEE transactions on knowledge and data engineering, Vol. 26, 7 (2013), 1763-1777."},{"key":"e_1_3_2_1_25_1","volume-title":"Proceedings of the 38th International Conference on Machine Learning (ICML).","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pam Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. In Proceedings of the 38th International Conference on Machine Learning (ICML)."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.676"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1514"},{"key":"e_1_3_2_1_28_1","volume-title":"Proceedings of the 24th International Conference on World Wide Web (WWW). 43-53","author":"Tang Jiliang","year":"2015","unstructured":"Jiliang Tang, Shiyu Chang, Charu Aggarwal, and Huan Liu. 2015. What contributes to a post's popularity? Content or users?. In Proceedings of the 24th International Conference on World Wide Web (WWW). 43-53."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3698261"},{"key":"e_1_3_2_1_30_1","volume-title":"Videomae: Masked autoencoders are data-efficient learners for self-supervised video pre-training. Advances in neural information processing systems","author":"Tong Zhan","year":"2022","unstructured":"Zhan Tong, Yibing Song, Jue Wang, and Limin Wang. 2022. Videomae: Masked autoencoders are data-efficient learners for self-supervised video pre-training. Advances in neural information processing systems, Vol. 35 (2022), 10078-10093."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3688999"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3688995"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3343031.3356084"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3613853"},{"key":"e_1_3_2_1_35_1","volume-title":"Proceedings of the 2016 ACM Multimedia Conference (MM). 501-510","author":"Wu Hao","year":"2016","unstructured":"Hao Wu, Bao Zhang, Jie Yang, and Fan Meng. 2016. Decomposing Popularity Evolution of Social Media Content Over Time. In Proceedings of the 2016 ACM Multimedia Conference (MM). 501-510."},{"key":"e_1_3_2_1_36_1","volume-title":"Proceedings of the ACM Multimedia Conference (MM).","author":"Wu Jianfeng","year":"2023","unstructured":"Jianfeng Wu, Wensheng Zhang, Shuhui Liu, et al., 2023b. SMP Challenge: Predicting the Popularity of Social Media Posts. In Proceedings of the ACM Multimedia Conference (MM)."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-00764-5_2"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3416274"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1121\/1.4984122"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1121\/10.0019937"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1121\/1.5111059"},{"key":"e_1_3_2_1_42_1","first-page":"2345","article-title":"Temporal Modeling for Short-Form Social Media Popularity Dynamics","volume":"27","author":"Xu Panpan","year":"2025","unstructured":"Panpan Xu, Jian Yang, and Heng-Tao Shen. 2025. Temporal Modeling for Short-Form Social Media Popularity Dynamics. IEEE Transactions on Multimedia, Vol. 27, 4 (2025), 2345-2358.","journal-title":"IEEE Transactions on Multimedia"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3416274"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10462-022-10283-5"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2025.3605964"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2024.112513"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.3008832"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3763762","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:44:06Z","timestamp":1765309446000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3763762"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":47,"alternative-id":["10.1145\/3746027.3763762","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3763762","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}