{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,4]],"date-time":"2026-05-04T05:43:35Z","timestamp":1777873415916,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":40,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100006374","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62172443"],"award-info":[{"award-number":["62172443"]}],"id":[{"id":"10.13039\/501100006374","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,8,3]]},"DOI":"10.1145\/3711896.3737404","type":"proceedings-article","created":{"date-parts":[[2025,8,3]],"date-time":"2025-08-03T20:52:41Z","timestamp":1754254361000},"page":"5842-5853","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["When Graph Meets Multimodal: Benchmarking and Meditating on Multimodal Attributed Graph Learning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-7631-9375","authenticated-orcid":false,"given":"Hao","family":"Yan","sequence":"first","affiliation":[{"name":"Central South University, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8179-7503","authenticated-orcid":false,"given":"Chaozhuo","family":"Li","sequence":"additional","affiliation":[{"name":"Microsoft Research Asia, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-6714-3476","authenticated-orcid":false,"given":"Jun","family":"Yin","sequence":"additional","affiliation":[{"name":"Central South University, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-3767-6247","authenticated-orcid":false,"given":"Zhigang","family":"Yu","sequence":"additional","affiliation":[{"name":"Central South University, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5533-6455","authenticated-orcid":false,"given":"Weihao","family":"Han","sequence":"additional","affiliation":[{"name":"Microsoft AI, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5615-9545","authenticated-orcid":false,"given":"Mingzheng","family":"Li","sequence":"additional","affiliation":[{"name":"Microsoft AI, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-6573-0658","authenticated-orcid":false,"given":"Zhengxin","family":"Zeng","sequence":"additional","affiliation":[{"name":"Microsoft AI, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-5027-7478","authenticated-orcid":false,"given":"Hao","family":"Sun","sequence":"additional","affiliation":[{"name":"Microsoft AI, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3615-4859","authenticated-orcid":false,"given":"Senzhang","family":"Wang","sequence":"additional","affiliation":[{"name":"Central South University, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,8,3]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3460231.3474255"},{"key":"e_1_3_2_2_2_1","volume-title":"Proceedings of the 35th Annual Conference on Neural Information Processing Systems.","author":"Desai Karan","year":"2021","unstructured":"Karan Desai, Gaurav Kaul, Zubin Aysola, and Justin Johnson. 2021. RedCaps: Web-curated image-text data created by the people, for the people. In Proceedings of the 35th Annual Conference on Neural Information Processing Systems."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2308.02565"},{"key":"e_1_3_2_2_4_1","volume-title":"Proceedings of the 34th Annual Conference on Neural Information Processing Systems.","author":"Freitas Scott","year":"2020","unstructured":"Scott Freitas, Yuxiao Dong, Joshua Neil, and Duen Horng Chau. 2020. A Large-Scale Database for Graph Representation Learning. In Proceedings of the 34th Annual Conference on Neural Information Processing Systems."},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","unstructured":"Yunfan Gao Yun Xiong Xinyu Gao Kangxiang Jia Jinliu Pan Yuxi Bi Yi Dai Jiawei Sun and Haofen Wang. 2023. Retrieval-Augmented Generation for Large Language Models: A Survey. arXiv:2312.10997(2023). https:\/\/doi.org\/10.48550\/arXiv.2312.10997","DOI":"10.48550\/arXiv.2312.10997"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/2843948"},{"key":"e_1_3_2_2_7_1","volume-title":"Proceedings of the 31th Annual Conference on Neural Information Processing Systems.","author":"Hamilton Will","year":"2017","unstructured":"Will Hamilton, Zhitao Ying, and Jure Leskovec. 2017. Inductive representation learning on large graphs. In Proceedings of the 31th Annual Conference on Neural Information Processing Systems."},{"key":"e_1_3_2_2_8_1","volume-title":"Proceedings of the 12th International Conference on Learning Representations.","author":"He Xiaoxin","year":"2024","unstructured":"Xiaoxin He, Xavier Bresson, Thomas Laurent, Adam Perold, Yann LeCun, and Bryan Hooi. 2024. Harnessing Explanations: LLM-to-LM Interpreter for Enhanced Text-Attributed Graph Representation Learning. In Proceedings of the 12th International Conference on Learning Representations."},{"key":"e_1_3_2_2_9_1","volume-title":"Proceedings of the 34th Annual Conference on Neural Information Processing Systems.","author":"Hu Weihua","year":"2020","unstructured":"Weihua Hu, Matthias Fey, Marinka Zitnik, Yuxiao Dong, Hongyu Ren, Bowen Liu, Michele Catasta, and Jure Leskovec. 2020. Open graph benchmark: Datasets for machine learning on graphs. In Proceedings of the 34th Annual Conference on Neural Information Processing Systems."},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","unstructured":"Xuanwen Huang Kaiqiao Han Dezheng Bao Quanjin Tao Zhisheng Zhang Yang Yang and Qi Zhu. 2023. Prompt-based Node Feature Extractor for Few-shot Learning on Text-Attributed Graphs. arXiv:2309.02848(2023). https:\/\/doi.org\/10.48550\/arXiv.2309.02848","DOI":"10.48550\/arXiv.2309.02848"},{"key":"e_1_3_2_2_11_1","volume-title":"Proceedings of the 5th International Conference on Learning Representations.","author":"Thomas","unstructured":"Thomas N. Kipf and Max Welling. 2017. Semi-supervised classification with graph convolutional networks. In Proceedings of the 5th International Conference on Learning Representations."},{"key":"e_1_3_2_2_12_1","volume-title":"Proceedings of the 38th International Conference on Machine Learning.","author":"Li Guohao","year":"2021","unstructured":"Guohao Li, Matthias M\u00fcller, Bernard Ghanem, and Vladlen Koltun. 2021. Training Graph Neural Networks with 1000 Layers. In Proceedings of the 38th International Conference on Machine Learning."},{"key":"e_1_3_2_2_13_1","volume-title":"Proceedings of the 33th International Joint Conference on Artificial Intelligence.","author":"Li Yuhan","year":"2023","unstructured":"Yuhan Li, Zhixun Li, Peisong Wang, Jia Li, Xiangguo Sun, Hong Cheng, and Jeffrey Xu Yu. 2023. A Survey of Graph Meets Large Language Model: Progress and Future Directions. In Proceedings of the 33th International Joint Conference on Artificial Intelligence."},{"key":"e_1_3_2_2_14_1","volume-title":"Proceedings of the 38th Annual Conference on Neural Information Processing Systems.","author":"Li Yuhan","year":"2024","unstructured":"Yuhan Li, Peisong Wang, Xiao Zhu, Aochuan Chen, Haiyun Jiang, Deng Cai, Victor Wai Kin Chan, and Jia Li. 2024b. GLBench: A Comprehensive Benchmark for Graph with Large Language Models. In Proceedings of the 38th Annual Conference on Neural Information Processing Systems."},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","unstructured":"Yanwei Li Yuechen Zhang Chengyao Wang Zhisheng Zhong Yixin Chen Ruihang Chu Shaoteng Liu and Jiaya Jia. 2024c. Mini-Gemini: Mining the Potential of Multi-modality Vision Language Models. arXiv:2403.18814(2024). https:\/\/doi.org\/10.48550\/arXiv.2403.18814","DOI":"10.48550\/arXiv.2403.18814"},{"key":"e_1_3_2_2_16_1","volume-title":"Proceedings of the 38th Annual Conference on Neural Information Processing Systems.","author":"Li Zhuofeng","year":"2024","unstructured":"Zhuofeng Li, Zixing Gou, Xiangnan Zhang, Zhongyuan Liu, Sirui Li, Yuntong Hu, Chen Ling, Zheng Zhang, and Liang Zhao. 2024a. TEG-DB: A Comprehensive Dataset and Benchmark of Textual-Edge Graphs. In Proceedings of the 38th Annual Conference on Neural Information Processing Systems."},{"key":"e_1_3_2_2_17_1","volume-title":"Proceedings of the 37th Annual Conference on Neural Information Processing Systems.","author":"Liu Haotian","year":"2023","unstructured":"Haotian Liu, Chunyuan Li, Qingyang Wu, and Yong Jae Lee. 2023. Visual instruction tuning. In Proceedings of the 37th Annual Conference on Neural Information Processing Systems."},{"key":"e_1_3_2_2_18_1","volume-title":"Roberta: A robustly optimized bert pretraining approach. arXiv:1907.11692(2019)","author":"Liu Yinhan","year":"2019","unstructured":"Yinhan Liu, Myle Ott, Naman Goyal, Jingfei Du, Mandar Joshi, Danqi Chen, Omer Levy, Mike Lewis, Luke Zettlemoyer, and Veselin Stoyanov. 2019. Roberta: A robustly optimized bert pretraining approach. arXiv:1907.11692(2019). http:\/\/arxiv.org\/abs\/1907.11692"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01170"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"e_1_3_2_2_21_1","volume-title":"Proceedings of the 16th ACM International Conference on Web Search and Data Mining.","author":"Naseem Usman","unstructured":"Usman Naseem, Jinman Kim, Matloob Khushi, and Adam G. Dunn. 2023. A Multimodal Framework for the Identification of Vaccine Critical Memes on Twitter. In Proceedings of the 16th ACM International Conference on Web Search and Data Mining."},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1018"},{"key":"e_1_3_2_2_23_1","volume-title":"Proceedings of the 38th International Conference on Machine Learning.","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. In Proceedings of the 38th International Conference on Machine Learning."},{"key":"e_1_3_2_2_24_1","unstructured":"Oleksandr Shchur Maximilian Mumme Aleksandar Bojchevski and Stephan G\u00fcnnemann. 2018. Pitfalls of graph neural network evaluation. arXiv:1811.05868(2018). http:\/\/arxiv.org\/abs\/1811.05868"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657775"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ipm.2020.102277"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","unstructured":"Hugo Touvron Louis Martin Kevin Stone Peter Albert Amjad Almahairi Yasmine Babaei Nikolay Bashlykov Soumya Batra Prajjwal Bhargava Shruti Bhosale et al. 2023. Llama 2: Open Foundation and Fine-Tuned Chat Models. arXiv:2307.09288(2023). https:\/\/doi.org\/10.48550\/arXiv.2307.09288","DOI":"10.48550\/arXiv.2307.09288"},{"key":"e_1_3_2_2_28_1","volume-title":"Proceedings of the 6th International Conference on Learning Representations.","author":"Velickovic Petar","year":"2018","unstructured":"Petar Velickovic, Guillem Cucurull, Arantxa Casanova, Adriana Romero, Pietro Li\u00f2, and Yoshua Bengio. 2018. Graph attention networks. In Proceedings of the 6th International Conference on Learning Representations."},{"key":"e_1_3_2_2_29_1","unstructured":"Minjie Wang Da Zheng Zihao Ye Quan Gan Mufei Li Xiang Song Jinjing Zhou Chao Ma Lingfan Yu Yu Gai et al. 2019. Deep graph library: A graph-centric highly-performant package for graph neural networks. arXiv:1909.01315(2019). http:\/\/arxiv.org\/abs\/1909.01315"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","unstructured":"Peng Wang Shuai Bai Sinan Tan Shijie Wang Zhihao Fan Jinze Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge et al. 2024. Qwen2-vl: Enhancing vision-language model's perception of the world at any resolution. arXiv:2409.12191(2024). https:\/\/doi.org\/10.48550\/arXiv.2409.12191","DOI":"10.48550\/arXiv.2409.12191"},{"key":"e_1_3_2_2_31_1","volume-title":"So Kweon, and Saining Xie. 2023. Convnext v2: Co-designing and scaling convnets with masked autoencoders. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition.","author":"Woo Sanghyun","unstructured":"Sanghyun Woo, Shoubhik Debnath, Ronghang Hu, Xinlei Chen, Zhuang Liu, In So Kweon, and Saining Xie. 2023. Convnext v2: Co-designing and scaling convnets with masked autoencoders. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition."},{"key":"e_1_3_2_2_32_1","volume-title":"Proceedings of the 37th Annual Conference on Neural Information Processing Systems.","author":"Yan Hao","year":"2023","unstructured":"Hao Yan, Chaozhuo Li, Ruosong Long, Chao Yan, Jianan Zhao, Wenwen Zhuang, Jun Yin, Peiyan Zhang, Weihao Han, Hao Sun, Weiwei Deng, Qi Zhang, Lichao Sun, Xing Xie, and Senzhang Wang. 2023a. A Comprehensive Study on Text-attributed Graphs: Benchmarking and Rethinking. In Proceedings of the 37th Annual Conference on Neural Information Processing Systems."},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3701551.3703571"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-43415-0_41"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3696410.3714866"},{"key":"e_1_3_2_2_36_1","volume-title":"Proceedings of the 8th International Conference on Learning Representations.","author":"Zeng Hanqing","year":"2020","unstructured":"Hanqing Zeng, Hongkuan Zhou, Ajitesh Srivastava, Rajgopal Kannan, and Viktor Prasanna. 2020. GraphSAINT: Graph Sampling Based Inductive Learning Method. In Proceedings of the 8th International Conference on Learning Representations."},{"key":"e_1_3_2_2_37_1","volume-title":"Proceedings of the 38th Annual Conference on Neural Information Processing Systems.","author":"Zhang Jiasheng","year":"2024","unstructured":"Jiasheng Zhang, Jialin Chen, Menglin Yang, Aosong Feng, Shuang Liang, Jie Shao, and Rex Ying. 2024. DTGB: A Comprehensive Benchmark for Dynamic Text-Attributed Graphs. In Proceedings of the 38th Annual Conference on Neural Information Processing Systems."},{"key":"e_1_3_2_2_38_1","volume-title":"Proceedings of the 11th International Conference on Learning Representations.","author":"Zhao Jianan","year":"2023","unstructured":"Jianan Zhao, Meng Qu, Chaozhuo Li, Hao Yan, Qian Liu, Rui Li, Xing Xie, and Jian Tang. 2023. Learning on Large-scale Text-attributed Graphs via Variational Inference. In Proceedings of the 11th International Conference on Learning Representations."},{"key":"e_1_3_2_2_39_1","volume-title":"Proceedings of the 35th Annual Conference on Neural Information Processing Systems.","author":"Zheng Qinkai","year":"2021","unstructured":"Qinkai Zheng, Xu Zou, Yuxiao Dong, Yukuo Cen, Da Yin, Jiarong Xu, Yang Yang, and Jie Tang. 2021. Graph Robustness Benchmark: Benchmarking the Adversarial Robustness of Graph Machine Learning. In Proceedings of the 35th Annual Conference on Neural Information Processing Systems."},{"key":"e_1_3_2_2_40_1","volume-title":"Proceedings of the 12th International Conference on Learning Representations.","author":"Zhu Deyao","year":"2024","unstructured":"Deyao Zhu, Jun Chen, Xiaoqian Shen, Xiang Li, and Mohamed Elhoseiny. 2024. MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models. In Proceedings of the 12th International Conference on Learning Representations."}],"event":{"name":"KDD '25: The 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Toronto ON Canada","acronym":"KDD '25","sponsor":["SIGKDD ACM Special Interest Group on Knowledge Discovery in Data","SIGMOD ACM Special Interest Group on Management of Data"]},"container-title":["Proceedings of the 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.2"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3711896.3737404","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T17:56:00Z","timestamp":1777571760000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3711896.3737404"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8,3]]},"references-count":40,"alternative-id":["10.1145\/3711896.3737404","10.1145\/3711896"],"URL":"https:\/\/doi.org\/10.1145\/3711896.3737404","relation":{},"subject":[],"published":{"date-parts":[[2025,8,3]]},"assertion":[{"value":"2025-08-03","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}