{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T03:08:08Z","timestamp":1784603288582,"version":"3.55.0"},"reference-count":120,"publisher":"Springer Science and Business Media LLC","issue":"11","license":[{"start":{"date-parts":[[2025,10,27]],"date-time":"2025-10-27T00:00:00Z","timestamp":1761523200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,10,27]],"date-time":"2025-10-27T00:00:00Z","timestamp":1761523200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Sci. China Inf. Sci."],"published-print":{"date-parts":[[2025,11]]},"DOI":"10.1007\/s11432-025-4627-3","type":"journal-article","created":{"date-parts":[[2025,10,30]],"date-time":"2025-10-30T09:34:21Z","timestamp":1761816861000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Towards multimodal graph large language model"],"prefix":"10.1007","volume":"68","author":[{"given":"Xin","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zeyang","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Linxin","family":"Xiao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haibo","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chendi","family":"Ge","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wenwu","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"4627_CR1","first-page":"327","volume-title":"Proceedings of the 12th International Conference on Affective Computing and Intelligent Interaction (ACII)","author":"S Bhattacharyya","year":"2024","unstructured":"Bhattacharyya S, Yang S, Wang J Z. A heterogeneous multimodal graph learning framework for recognizing user emotions in social networks. In: Proceedings of the 12th International Conference on Affective Computing and Intelligent Interaction (ACII), 2024. 327\u2013336"},{"key":"4627_CR2","first-page":"149","volume-title":"Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics","author":"X Wang","year":"2023","unstructured":"Wang X, Wang C, Li L, et al. Fashionklip: enhancing e-commerce image-text retrieval with fashion multi-modal conceptual knowledge graph. In: Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics, 2023. 149\u2013158"},{"key":"4627_CR3","doi-asserted-by":"publisher","first-page":"103303","DOI":"10.1016\/j.media.2024.103303","volume":"97","author":"N Marini","year":"2024","unstructured":"Marini N, Marchesin S, Wodzinski M, et al. Multimodal representations of biomedical knowledge from limited training whole slide images and reports using deep learning. Med Image Anal, 2024, 97: 103303","journal-title":"Med Image Anal"},{"key":"4627_CR4","first-page":"56878","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"Y Ye","year":"2024","unstructured":"Ye Y, Ren J, Wang S, et al. Construction and application of materials knowledge graph in multidisciplinary materials science via large language model. In: Proceedings of Advances in Neural Information Processing Systems, 2024. 56878\u201356897"},{"key":"4627_CR5","doi-asserted-by":"publisher","first-page":"340","DOI":"10.1038\/s42256-023-00624-6","volume":"5","author":"Y Ektefaie","year":"2023","unstructured":"Ektefaie Y, Dasoulas G, Noori A, et al. Multimodal learning with graphs. Nat Mach Intell, 2023, 5: 340\u2013350","journal-title":"Nat Mach Intell"},{"key":"4627_CR6","unstructured":"Peng C, He J, Xia F. Learning on multimodal graphs: a survey. ArXiv:2402.05322"},{"key":"4627_CR7","unstructured":"Chen R, Zhao T, Jaiswal A, et al. LLAGA: large language and graph assistant. ArXiv:2402.08170"},{"key":"4627_CR8","unstructured":"Kong L, Feng J, Liu H, et al. Gofa: a generative one-for-all model for joint graph language modeling. ArXiv:2407.09709"},{"key":"4627_CR9","first-page":"8617","volume-title":"Proceedings of Findings of the Association for Computational Linguistics","author":"S Jiang","year":"2025","unstructured":"Jiang S, Liang J, Wang J, et al. From specific-MLLMs to OMNI-MLLMs: a survey on MLLMs aligned with multi-modalities. In: Proceedings of Findings of the Association for Computational Linguistics, 2025. 8617\u20138652"},{"key":"4627_CR10","unstructured":"Zhu J, Zhou Y, Qian S, et al. Mosaic of modalities: a comprehensive benchmark for multimodal graph learning. ArXiv:2406.16321"},{"key":"4627_CR11","doi-asserted-by":"publisher","first-page":"130","DOI":"10.1007\/978-3-030-41407-8_9","volume-title":"Proceedings of Semantic Technology. Springer International Publishing","author":"M Wang","year":"2020","unstructured":"Wang M, Qi G, Wang H, et al. Richpedia: a comprehensive multi-modal knowledge graph. In: Proceedings of Semantic Technology. Springer International Publishing, 2020. 130\u2013145"},{"key":"4627_CR12","first-page":"6693","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"D A Hudson","year":"2019","unstructured":"Hudson D A, Manning C D. GQA: a new dataset for real-world visual reasoning and compositional question answering. In: Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), 2019. 6693\u20136702"},{"key":"4627_CR13","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3674501","volume":"56","author":"F Zhao","year":"2024","unstructured":"Zhao F, Zhang C, Geng B. Deep multimodal data fusion. ACM Comput Surv, 2024, 56: 1\u201336","journal-title":"ACM Comput Surv"},{"key":"4627_CR14","doi-asserted-by":"publisher","first-page":"1437","DOI":"10.1145\/3343031.3351034","volume-title":"Proceedings of the 27th ACM International Conference on Multimedia","author":"Y Wei","year":"2019","unstructured":"Wei Y, Wang X, Nie L, et al. MMGCN: multi-modal graph convolution network for personalized recommendation of microvideo. In: Proceedings of the 27th ACM International Conference on Multimedia, 2019. 1437\u20131445"},{"key":"4627_CR15","doi-asserted-by":"publisher","first-page":"354","DOI":"10.1109\/TMI.2022.3187141","volume":"42","author":"X Song","year":"2023","unstructured":"Song X, Zhou F, Frangi A F, et al. Multicenter and multichannel pooling GCN for early AD diagnosis based on dual-modality fused brain network. IEEE Trans Med Imag, 2023, 42: 354\u2013367","journal-title":"IEEE Trans Med Imag"},{"key":"4627_CR16","first-page":"3376","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence","author":"Y Zeng","year":"2023","unstructured":"Zeng Y, Jin Q, Bao T, et al. Multi-modal knowledge hypergraph for diverse image retrieval. In: Proceedings of the AAAI Conference on Artificial Intelligence, 2023. 3376\u20133383"},{"key":"4627_CR17","first-page":"3473","volume-title":"Proceedings of the 31st International Joint Conference on Artificial Intelligence","author":"Y Tian","year":"2022","unstructured":"Tian Y, Zhang C, Guo Z, et al. Recipe2vec: multi-modal recipe representation learning with graph neural networks. In: Proceedings of the 31st International Joint Conference on Artificial Intelligence, 2022. 3473\u20133479"},{"key":"4627_CR18","doi-asserted-by":"publisher","first-page":"189","DOI":"10.18653\/v1\/2023.eacl-main.15","volume-title":"Proceedings of the 17th Conference of the European Chapter of the Association for Computational Linguistics","author":"X He","year":"2023","unstructured":"He X, Wang X. Multimodal graph transformer for multimodal question answering. In: Proceedings of the 17th Conference of the European Chapter of the Association for Computational Linguistics, 2023. 189\u2013200"},{"key":"4627_CR19","doi-asserted-by":"publisher","first-page":"456","DOI":"10.1109\/TMI.2022.3222093","volume":"42","author":"H Cai","year":"2023","unstructured":"Cai H, Gao Y, Liu M. Graph transformer geometric learning of brain networks using multimodal MR images for brain age estimation. IEEE Trans Med Imag, 2023, 42: 456\u2013466","journal-title":"IEEE Trans Med Imag"},{"key":"4627_CR20","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"H Wang","year":"2024","unstructured":"Wang H, Feng S, He T, et al. Can language models solve graph problems in natural language? In: Proceedings of Advances in Neural Information Processing Systems, 2024"},{"key":"4627_CR21","unstructured":"Guo J, Du L, Liu H, et al. GPT4graph: can large language models understand graph structured data? An empirical evaluation and benchmarking. ArXiv:2305.15066"},{"key":"4627_CR22","first-page":"5850","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"H Zhao","year":"2023","unstructured":"Zhao H, Liu S, Chang M, et al. GIMLET: a unified graph-text model for instruction-based molecule zero-shot learning. In: Proceedings of Advances in Neural Information Processing Systems, 2023. 5850\u20135887"},{"key":"4627_CR23","doi-asserted-by":"publisher","first-page":"108073","DOI":"10.1016\/j.compbiomed.2024.108073","volume":"171","author":"P Liu","year":"2024","unstructured":"Liu P, Ren Y, Tao J, et al. GIT-Mol: a multi-modal large language model for molecular science with graph, image, and text. Comput Biol Med, 2024, 171: 108073","journal-title":"Comput Biol Med"},{"key":"4627_CR24","unstructured":"Liu C, Wu B. Evaluating large language models on graphs: performance insights and comparative analysis. ArXiv:2308.11224"},{"key":"4627_CR25","unstructured":"Fatemi B, Halcrow J, Perozzi B. Talk like a graph: encoding graphs for large language models. ArXiv:2310.04560"},{"key":"4627_CR26","unstructured":"Hu Y, Zhang Z, Zhao L. Beyond text: a deep dive into large language models\u2019 ability on understanding graph data. ArXiv:2310.04944"},{"key":"4627_CR27","doi-asserted-by":"publisher","first-page":"8227","DOI":"10.1609\/aaai.v38i8.28663","volume":"38","author":"J Cai","year":"2024","unstructured":"Cai J, Wang X, Li H, et al. Multimodal graph neural architecture search under distribution shifts. AAAI, 2024, 38: 8227\u20138235","journal-title":"AAAI"},{"key":"4627_CR28","doi-asserted-by":"publisher","first-page":"i457","DOI":"10.1093\/bioinformatics\/bty294","volume":"34","author":"M Zitnik","year":"2018","unstructured":"Zitnik M, Agrawal M, Leskovec J. Modeling polypharmacy side effects with graph convolutional networks. Bioinformatics, 2018, 34: i457\u2013i466","journal-title":"Bioinformatics"},{"key":"4627_CR29","doi-asserted-by":"publisher","first-page":"4","DOI":"10.1109\/TNNLS.2020.2978386","volume":"32","author":"Z Wu","year":"2020","unstructured":"Wu Z, Pan S, Chen F, et al. A comprehensive survey on graph neural networks. IEEE Trans Neural Netw Learn Syst, 2020, 32: 4\u201324","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"key":"4627_CR30","doi-asserted-by":"publisher","first-page":"176","DOI":"10.1162\/tacl_a_00360","volume":"9","author":"X Wang","year":"2021","unstructured":"Wang X, Gao T, Zhu Z, et al. KEPLER: a unified model for knowledge embedding and pre-trained language representation. Trans Assoc Comput Linguist, 2021, 9: 176\u2013194","journal-title":"Trans Assoc Comput Linguist"},{"key":"4627_CR31","first-page":"1929","volume":"15","author":"N Srivastava","year":"2014","unstructured":"Srivastava N, Hinton G, Krizhevsky A, et al. Dropout: a simple way to prevent neural networks from overfitting. J Mach Learn Res, 2014, 15: 1929\u20131958","journal-title":"J Mach Learn Res"},{"key":"4627_CR32","volume-title":"Proceedings of the 41st International Conference on Machine Learning","author":"H Mao","year":"2024","unstructured":"Mao H, Chen Z, Tang W, et al. Position: graph foundation models are already here. In: Proceedings of the 41st International Conference on Machine Learning, 2024"},{"key":"4627_CR33","volume-title":"Proceedings of the 12th International Conference on Learning Representations","author":"H Liu","year":"2024","unstructured":"Liu H, Feng J, Kong L, et al. One for all: towards training one graph model for all classification tasks. In: Proceedings of the 12th International Conference on Learning Representations, 2024"},{"key":"4627_CR34","volume-title":"Proceedings of the 12th International Conference on Learning Representations","author":"M Galkin","year":"2024","unstructured":"Galkin M, Yuan X, Mostafa H, et al. Towards foundation models for knowledge graph reasoning. In: Proceedings of the 12th International Conference on Learning Representations, 2024"},{"key":"4627_CR35","unstructured":"Zhu Y, Shi H, Wang X, et al. GraphCLIP: enhancing transferability in graph foundation models for text-attributed graphs. ArXiv:2410.10329"},{"key":"4627_CR36","first-page":"1955","volume-title":"Proceedings of Findings of the Association for Computational Linguistics","author":"R Ye","year":"2024","unstructured":"Ye R, Zhang C, Wang R, et al. Language is all a graph needs. In: Proceedings of Findings of the Association for Computational Linguistics, 2024. 1955\u20131973"},{"key":"4627_CR37","first-page":"354","volume-title":"Proceedings of the 31st International Conference on Computational Linguistics","author":"H Cao","year":"2025","unstructured":"Cao H, Liu Z, Lu X, et al. Instructmol: multi-modal integration for building a versatile and reliable molecular assistant in drug discovery. In: Proceedings of the 31st International Conference on Computational Linguistics, 2025. 354\u2013379"},{"key":"4627_CR38","first-page":"34892","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"H Liu","year":"2023","unstructured":"Liu H, Li C, Wu Q, et al. Visual instruction tuning. In: Proceedings of Advances in Neural Information Processing Systems, 2023. 34892\u201334916"},{"key":"4627_CR39","unstructured":"Yoon M, Koh J Y, Hooi B, et al. Multimodal graph learning for generative tasks. ArXiv:2310.07478"},{"key":"4627_CR40","doi-asserted-by":"publisher","first-page":"2120","DOI":"10.1145\/3580305.3599256","volume-title":"Proceedings of the 29th ACM SIGKDD Conference on Knowledge Discovery and Data Mining","author":"X Sun","year":"2023","unstructured":"Sun X, Cheng H, Li J, et al. All in one: multi-task prompting for graph neural networks. In: Proceedings of the 29th ACM SIGKDD Conference on Knowledge Discovery and Data Mining, 2023. 2120\u20132131"},{"key":"4627_CR41","doi-asserted-by":"publisher","first-page":"478","DOI":"10.1109\/JSTSP.2020.2987728","volume":"14","author":"C Zhang","year":"2020","unstructured":"Zhang C, Yang Z, He X, et al. Multimodal intelligence: representation learning, information fusion, and applications. IEEE J Sel Top Signal Process, 2020, 14: 478\u2013493","journal-title":"IEEE J Sel Top Signal Process"},{"key":"4627_CR42","doi-asserted-by":"publisher","first-page":"1150","DOI":"10.1145\/3394486.3403168","volume-title":"Proceedings of the 26th ACM SIGKDD International Conference on Knowledge Discovery & Data Mining","author":"J Qiu","year":"2020","unstructured":"Qiu J, Chen Q, Dong Y, et al. GCC: graph contrastive coding for graph neural network pre-training. In: Proceedings of the 26th ACM SIGKDD International Conference on Knowledge Discovery & Data Mining, 2020. 1150\u20131160"},{"key":"4627_CR43","doi-asserted-by":"publisher","first-page":"1717","DOI":"10.1145\/3534678.3539249","volume-title":"Proceedings of the 28th ACM SIGKDD Conference on Knowledge Discovery and Data Mining","author":"M Sun","year":"2022","unstructured":"Sun M, Zhou K, He X, et al. GPPT: graph pre-training and prompt tuning to generalize graph neural networks. In: Proceedings of the 28th ACM SIGKDD Conference on Knowledge Discovery and Data Mining, 2022. 1717\u20131727"},{"key":"4627_CR44","first-page":"417","volume-title":"Proceedings of the ACM Web Conference","author":"Z Liu","year":"2023","unstructured":"Liu Z, Yu X, Fang Y, et al. Graphprompt: unifying pre-training and downstream tasks for graph neural networks. In: Proceedings of the ACM Web Conference, 2023. 417\u2013428"},{"key":"4627_CR45","first-page":"4328","volume-title":"Proceedings of the ACM on Web Conference","author":"Y Yan","year":"2024","unstructured":"Yan Y, Zhang P, Fang Z, et al. Inductive graph alignment prompt: bridging the GAP between graph pre-training and inductive fine-tuning from spectral perspective. In: Proceedings of the ACM on Web Conference, 2024. 4328\u20134339"},{"key":"4627_CR46","first-page":"515","volume-title":"Proceedings of the ACM on Web Conference","author":"X Yu","year":"2024","unstructured":"Yu X, Zhou C, Fang Y, et al. Multigprompt for multi-task pre-training and prompting on graphs. In: Proceedings of the ACM on Web Conference, 2024. 515\u2013526"},{"key":"4627_CR47","doi-asserted-by":"publisher","first-page":"4443","DOI":"10.1145\/3637528.3671913","volume-title":"Proceedings of the 30th ACM SIGKDD Conference on Knowledge Discovery and Data Mining","author":"H Zhao","year":"2024","unstructured":"Zhao H, Chen A, Sun X, et al. All in one and one for all: a simple yet effective method towards cross-domain graph pretraining. In: Proceedings of the 30th ACM SIGKDD Conference on Knowledge Discovery and Data Mining, 2024. 4443\u20134454"},{"key":"4627_CR48","doi-asserted-by":"publisher","first-page":"18","DOI":"10.1109\/MSP.2017.2693418","volume":"34","author":"M M Bronstein","year":"2017","unstructured":"Bronstein M M, Bruna J, LeCun Y, et al. Geometric deep learning: going beyond Euclidean data. IEEE Signal Process Mag, 2017, 34: 18\u201342","journal-title":"IEEE Signal Process Mag"},{"key":"4627_CR49","first-page":"1877","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"T Brown","year":"2020","unstructured":"Brown T, Mann B, Ryder N, et al. Language models are few-shot learners. In: Proceedings of Advances in Neural Information Processing Systems, 2020. 1877\u20131901"},{"key":"4627_CR50","first-page":"12096","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"R Sun","year":"2022","unstructured":"Sun R, Dai H, Yu A W. Does GNN pretraining help molecular representation? In: Proceedings of Advances in Neural Information Processing Systems, 2022. 12096\u201312109"},{"key":"4627_CR51","first-page":"893","volume-title":"Proceedings of the ACM Web Conference","author":"X Huang","year":"2024","unstructured":"Huang X, Han K, Yang Y, et al. Can GNN be good adapter for LLMs? In: Proceedings of the ACM Web Conference, 2024. 893\u2013904"},{"key":"4627_CR52","first-page":"9459","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"P Lewis","year":"2020","unstructured":"Lewis P, Perez E, Piktus A, et al. Retrieval-augmented generation for knowledge-intensive NLP tasks. In: Proceedings of Advances in Neural Information Processing Systems, 2020. 9459\u20139474"},{"key":"4627_CR53","doi-asserted-by":"publisher","first-page":"491","DOI":"10.1145\/3626772.3657775","volume-title":"Proceedings of the 47th International ACM SIGIR Conference on Research and Development in Information Retrieval","author":"J Tang","year":"2024","unstructured":"Tang J, Yang Y, Wei W, et al. GraphGPT: graph instruction tuning for large language models. In: Proceedings of the 47th International ACM SIGIR Conference on Research and Development in Information Retrieval, 2024. 491\u2013500"},{"key":"4627_CR54","unstructured":"Yu S, Wang Y, Li R, et al. Graph2text or graph2token: a perspective of large language models for graph learning. ArXiv:2501.01124"},{"key":"4627_CR55","first-page":"10767","volume-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics","author":"J Lee","year":"2024","unstructured":"Lee J, Wang Y, Li J, et al. Multimodal reasoning with multimodal knowledge graph. In: Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics, 2024. 10767\u201310782"},{"key":"4627_CR56","unstructured":"Chai Z, Zhang T, Wu L, et al. GraphLLM: boosting graph reasoning ability of large language model. ArXiv:2310.05845"},{"key":"4627_CR57","unstructured":"Li R, Jiang H. Graph-to-vision: multi-graph understanding and reasoning using vision-language models. ArXiv:2503.21435"},{"key":"4627_CR58","unstructured":"Xia F, Li B, Weng Y, et al. Lingyi: medical conversational question answering system based on multi-modal knowledge graphs. ArXiv:2204.09220"},{"key":"4627_CR59","unstructured":"Huang Y, Shi L, Liu A, et al. Evaluating and enhancing large language models for conversational reasoning on knowledge graphs. ArXiv:2312.11282"},{"key":"4627_CR60","unstructured":"Liu M, Xu J. NLI4DB: a systematic review of natural language interfaces for databases. ArXiv:2503.02435"},{"key":"4627_CR61","first-page":"2417","volume-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics","author":"J Dong","year":"2024","unstructured":"Dong J, Zhang Q, Zhou H, et al. Modality-aware integration with large language models for knowledge-based visual question answering. In: Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics, 2024. 2417\u20132429"},{"key":"4627_CR62","unstructured":"Huang N, Deshpande Y R, Liu Y, et al. Endowing language models with multimodal knowledge graph representations. ArXiv:2206.13163"},{"key":"4627_CR63","first-page":"15534","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"S Chen","year":"2022","unstructured":"Chen S, Li B. Multi-modal dynamic graph transformer for visual grounding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022. 15534\u201315543"},{"key":"4627_CR64","unstructured":"Edge D, Trinh H, Cheng N, et al. From local to global: a graph RAG approach to query-focused summarization. ArXiv:2404.16130"},{"key":"4627_CR65","volume-title":"Proceedings of the 11th International Conference on Learning Representations","author":"A Creswell","year":"2023","unstructured":"Creswell A, Shanahan M, Higgins I. Selection-inference: exploiting large language models for interpretable logical reasoning. In: Proceedings of the 11th International Conference on Learning Representations, 2023"},{"key":"4627_CR66","volume-title":"Proceedings of the Conference on Empirical Methods in Natural Language Processing","author":"J Wei","year":"2023","unstructured":"Wei J, Hou L, Lampinen A K, et al. Symbol tuning improves in-context learning in language models. In: Proceedings of the Conference on Empirical Methods in Natural Language Processing, 2023"},{"key":"4627_CR67","volume-title":"Proceedings of International Conference on Learning Representations","author":"A Talmor","year":"2021","unstructured":"Talmor A, Yoran O, Catav A, et al. Multimodal{qa}: complex question answering over text, tables and images. In: Proceedings of International Conference on Learning Representations, 2021"},{"key":"4627_CR68","volume-title":"Proceedings of the 11th International Conference on Learning Representations","author":"N Zhang","year":"2023","unstructured":"Zhang N, Li L, Chen X, et al. Multimodal analogical reasoning over knowledge graphs. In: Proceedings of the 11th International Conference on Learning Representations, 2023"},{"key":"4627_CR69","unstructured":"Guo D, Cao C, Yuan F, et al. Can multimodal large language model think analogically? ArXiv:2411.01307"},{"key":"4627_CR70","first-page":"8748","volume-title":"Proceedings of International Conference on Machine Learning","author":"A Radford","year":"2021","unstructured":"Radford A, Kim J W, Hallacy C, et al. Learning transferable visual models from natural language supervision. In: Proceedings of International Conference on Machine Learning, 2021. 8748\u20138763"},{"key":"4627_CR71","first-page":"3235","volume-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics","author":"X Chen","year":"2024","unstructured":"Chen X, Wang C, Xue Y, et al. Unified hallucination detection for multimodal large language models. In: Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics, 2024. 3235\u20133252"},{"key":"4627_CR72","volume-title":"Proceedings of the 12th International Conference on Learning Representations","author":"X Liu","year":"2024","unstructured":"Liu X, LI R, Ji W, et al. Towards robust multi-modal reasoning via model selection. In: Proceedings of the 12th International Conference on Learning Representations, 2024"},{"key":"4627_CR73","volume-title":"Proceedings of ICLR Conference","author":"I Choi","year":"2025","unstructured":"Choi I, Yun S, Xin J, et al. Multimodal graph-LLM: leveraging graph-enhanced LLMs for multimodal healthcare predictions. In: Proceedings of ICLR Conference, 2025"},{"key":"4627_CR74","doi-asserted-by":"publisher","first-page":"904","DOI":"10.1145\/3477495.3531992","volume-title":"Proceedings of the 45th International ACM SIGIR Conference on Research and Development in Information Retrieval","author":"X Chen","year":"2022","unstructured":"Chen X, Zhang N, Li L, et al. Hybrid transformer with multi-level fusion for multimodal knowledge graph completion. In: Proceedings of the 45th International ACM SIGIR Conference on Research and Development in Information Retrieval, 2022. 904\u2013915"},{"key":"4627_CR75","unstructured":"Kaiser L, Gomez A N, Shazeer N, et al. One model to learn them all. ArXiv:1706.05137"},{"key":"4627_CR76","first-page":"17283","volume-title":"Proceedings of Advances in Neural Information Processing Systems (NeurIPS)","author":"M Zaheer","year":"2020","unstructured":"Zaheer M, Guruganesh G, et al. Big bird: transformers for longer sequences. In: Proceedings of Advances in Neural Information Processing Systems (NeurIPS), 2020. 17283\u201317297"},{"key":"4627_CR77","first-page":"4651","volume-title":"Proceedings of the 38th International Conference on Machine Learning (ICML)","author":"A Jaegle","year":"2021","unstructured":"Jaegle A, Gimeno F, et al. Perceiver: general perception with iterative attention. In: Proceedings of the 38th International Conference on Machine Learning (ICML), 2021. 4651\u20134664"},{"key":"4627_CR78","unstructured":"Tay Y, Dehghani M, Bahri D, et al. Efficient transformers: a survey. ArXiv:2009.06732"},{"key":"4627_CR79","first-page":"1024","volume-title":"Proceedings of Advances in Neural Information Processing Systems (NeurIPS)","author":"W L Hamilton","year":"2017","unstructured":"Hamilton W L, Ying R, Leskovec J. Inductive representation learning on large graphs. In: Proceedings of Advances in Neural Information Processing Systems (NeurIPS), 2017. 1024\u20131034"},{"key":"4627_CR80","volume-title":"Proceedings of International Conference on Learning Representations (ICLR)","author":"H Zeng","year":"2020","unstructured":"Zeng H, Zhou H, Srivastava A, et al. Graphsaint: graph sampling based inductive learning method. In: Proceedings of International Conference on Learning Representations (ICLR), 2020"},{"key":"4627_CR81","first-page":"4800","volume-title":"Proceedings of Advances in Neural Information Processing Systems (NeurIPS)","author":"R Ying","year":"2018","unstructured":"Ying R, You J, Morris C, et al. Hierarchical graph representation learning with differentiable pooling. In: Proceedings of Advances in Neural Information Processing Systems (NeurIPS), 2018. 4800\u20134810"},{"key":"4627_CR82","unstructured":"Zheng D, Song X, Yang C, et al. Distributed hybrid CPU and GPU training for graph neural networks on billion-scale graphs. ArXiv:2010.05337"},{"key":"4627_CR83","first-page":"41","volume-title":"Proceedings of the 26th International Conference on Machine Learning (ICML)","author":"Y Bengio","year":"2009","unstructured":"Bengio Y, Louradour J, Collobert R, et al. Curriculum learning. In: Proceedings of the 26th International Conference on Machine Learning (ICML), 2009. 41\u201348"},{"key":"4627_CR84","unstructured":"Hinton G, Vinyals O, Dean J. Distilling the knowledge in a neural network. ArXiv:1503.02531"},{"key":"4627_CR85","volume-title":"Proceedings of the 4th International Conference on Learning Representations (ICLR)","author":"S Han","year":"2016","unstructured":"Han S, Mao H, Dally W J. Deep compression: compressing deep neural networks with pruning, trained quantization and Huffman coding. In: Proceedings of the 4th International Conference on Learning Representations (ICLR), 2016"},{"key":"4627_CR86","first-page":"28877","volume-title":"Proceedings of Advances in Neural Information Processing Systems (NeurIPS)","author":"C Ying","year":"2021","unstructured":"Ying C, Cai T, Luo S, et al. Do transformers really perform bad for graph representation? In: Proceedings of Advances in Neural Information Processing Systems (NeurIPS), 2021. 28877\u201328888"},{"key":"4627_CR87","first-page":"615","volume-title":"Proceedings of the 22nd International Conference on Architectural Support for Programming Languages and Operating Systems (ASPLOS)","author":"Y Kang","year":"2017","unstructured":"Kang Y, Hauswald J, Gao C, et al. Neurosurgeon: collaborative intelligence between the cloud and mobile edge. In: Proceedings of the 22nd International Conference on Architectural Support for Programming Languages and Operating Systems (ASPLOS), 2017. 615\u2013629"},{"key":"4627_CR88","first-page":"104353","volume":"136","author":"Y Tao","year":"2025","unstructured":"Tao Y, Liu W, Chen J, et al. A graph-based multimodal data fusion framework for identifying urban functional zone. Int J Appl Earth Obs GeoInf, 2025, 136: 104353","journal-title":"Int J Appl Earth Obs GeoInf"},{"key":"4627_CR89","doi-asserted-by":"publisher","first-page":"291","DOI":"10.1016\/j.cell.2018.03.022","volume":"173","author":"K A Hoadley","year":"2018","unstructured":"Hoadley K A, Yau C, Hinoue T, et al. Cell-of-origin patterns dominate the molecular classification of 10000 tumors from 33 types of cancer. Cell, 2018, 173: 291\u2013304","journal-title":"Cell"},{"key":"4627_CR90","doi-asserted-by":"publisher","first-page":"61","DOI":"10.1038\/nature11412","volume":"490","author":"Cancer Genome Atlas Network","year":"2012","unstructured":"Cancer Genome Atlas Network. Comprehensive molecular portraits of human breast tumours. Nature, 2012, 490: 61\u201370","journal-title":"Nature"},{"key":"4627_CR91","unstructured":"Saifuddin K M, Ji S, Akbas E. HyperGCL: multi-modal graph contrastive learning via learnable hypergraph views. ArXiv:2502.13277"},{"key":"4627_CR92","first-page":"7314","volume-title":"Proceedings of Findings of the Association for Computational Linguistics","author":"J Lee","year":"2023","unstructured":"Lee J, Chung C, Lee H, et al. Vista: visual-textual knowledge graph representation learning. In: Proceedings of Findings of the Association for Computational Linguistics, 2023. 7314\u20137328"},{"key":"4627_CR93","doi-asserted-by":"publisher","first-page":"2391","DOI":"10.1145\/3581783.3612266","volume-title":"Proceedings of the 31st ACM International Conference on Multimedia","author":"X Wang","year":"2023","unstructured":"Wang X, Meng B, Chen H, et al. TIVA-KG: a multimodal knowledge graph with text, image, video and audio. In: Proceedings of the 31st ACM International Conference on Multimedia, 2023. 2391\u20132399"},{"key":"4627_CR94","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR)","author":"J Johnson","year":"2017","unstructured":"Johnson J, Hariharan B, van der Maaten L, et al. CLEVR: a diagnostic dataset for compositional language and elementary visual reasoning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 2017"},{"key":"4627_CR95","unstructured":"Alonso I, Salaberria A, Azkune G, et al. Vision-language models struggle to align entities across modalities. ArXiv:2503.03854"},{"key":"4627_CR96","first-page":"225","volume-title":"Proceedings of the 7th Joint Conference on Lexical and Computational Semantics","author":"H Mousselly-Sergieh","year":"2018","unstructured":"Mousselly-Sergieh H, Botschen T, Gurevych I, et al. A multimodal translation-based approach for knowledge graph representation learning. In: Proceedings of the 7th Joint Conference on Lexical and Computational Semantics, 2018. 225\u2013234"},{"key":"4627_CR97","first-page":"3570","volume-title":"Proceedings of the 29th International Conference on Computational Linguistics","author":"K Shirai","year":"2022","unstructured":"Shirai K, Hashimoto A, Nishimura T, et al. Visual recipe flow: a dataset for learning visual state changes of objects with recipe flows. In: Proceedings of the 29th International Conference on Computational Linguistics, 2022. 3570\u20133577"},{"key":"4627_CR98","doi-asserted-by":"publisher","first-page":"4871","DOI":"10.18653\/v1\/2020.acl-main.440","volume-title":"Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics","author":"A Lin","year":"2020","unstructured":"Lin A, Rao S, Celikyilmaz A, et al. A recipe for creating multimodal aligned datasets for sequential tasks. In: Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics, 2020. 4871\u20134884"},{"key":"4627_CR99","unstructured":"Fang Y, Jin B, Shen J, et al. GraphGPT-O: synergistic multimodal comprehension and generation on graphs. ArXiv:2502.11925"},{"key":"4627_CR100","unstructured":"Jin B, Pang Z, Guo B, et al. Instructg2I: synthesizing images from multimodal attributed graphs. ArXiv:2410.07157"},{"key":"4627_CR101","volume-title":"Proceedings of International Conference on Learning Representations (ICLR)","author":"T N Kipf","year":"2017","unstructured":"Kipf T N, Welling M. Semi-supervised classification with graph convolutional networks. In: Proceedings of International Conference on Learning Representations (ICLR), 2017"},{"key":"4627_CR102","volume-title":"Proceedings of International Conference on Learning Representations (ICLR)","author":"P Veli\u010dkovi\u2019c","year":"2018","unstructured":"Veli\u010dkovi\u2019c P, Cucurull G, Casanova A, et al. Graph attention networks. In: Proceedings of International Conference on Learning Representations (ICLR), 2018"},{"key":"4627_CR103","volume-title":"Proceedings of International Conference on Learning Representations (ICLR)","author":"K Xu","year":"2019","unstructured":"Xu K, Hu W, Leskovec J, et al. How powerful are graph neural networks? In: Proceedings of International Conference on Learning Representations (ICLR), 2019"},{"key":"4627_CR104","unstructured":"Zhang S, Sohrabizadeh A, Wan C, et al. A survey on graph neural network acceleration: algorithms, systems, and customized hardware. ArXiv:2306.14052"},{"key":"4627_CR105","doi-asserted-by":"publisher","first-page":"1011","DOI":"10.1007\/s11633-024-1510-8","volume":"21","author":"E Dai","year":"2024","unstructured":"Dai E, Zhao T, Zhu H, et al. A comprehensive survey on trustworthy graph neural networks: privacy, robustness, fairness, and explainability. Mach Intell Res, 2024, 21: 1011\u20131061","journal-title":"Mach Intell Res"},{"key":"4627_CR106","unstructured":"Han H, Wang Y, Shomer H, et al. Retrieval-augmented generation with graphs (graphrag). ArXiv:2501.00309"},{"key":"4627_CR107","unstructured":"OpenAI. GPT-4 technical report. ArXiv:2303.08774"},{"key":"4627_CR108","unstructured":"Grattafiori A, Dubey A, Jauhri A, et al. The llama 3 herd of models. ArXiv:2407.21783"},{"key":"4627_CR109","unstructured":"Yang A, Li A, Yang B, et al. QWEN3 technical report. ArXiv:2505.09388"},{"key":"4627_CR110","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"A Vaswani","year":"2017","unstructured":"Vaswani A, Shazeer N, Parmar N, et al. Attention is all you need. In: Proceedings of Advances in Neural Information Processing Systems, 2017"},{"key":"4627_CR111","volume-title":"GPT-4V(ision) system card","author":"OpenAI","year":"2023","unstructured":"OpenAI. GPT-4V(ision) system card. 2023. https:\/\/openai.com\/index\/gpt-4v-systemcard\/"},{"key":"4627_CR112","unstructured":"Bai S, Chen K, Liu X, et al. Qwen2. 5-vl technical report. ArXiv:2502.13923"},{"key":"4627_CR113","unstructured":"Wang W, Gao Z, Gu L, et al. InternVL3.5: advancing open-source multimodal models in versatility, reasoning, and efficiency. ArXiv:2508.18265"},{"key":"4627_CR114","unstructured":"Team V, Hong W, Yu W, et al. GLM-4.5V and GLM-4.1V-thinking: towards versatile multimodal reasoning with scalable reinforcement learning. ArXiv:2507.01006"},{"key":"4627_CR115","unstructured":"Hurst A, Lerer A, Goucher A P, et al. GPT-4O system card. ArXiv:2410.21276"},{"key":"4627_CR116","unstructured":"Comanici G, Bieber E, Schaekermann M, et al. Gemini 2.5: pushing the frontier with advanced reasoning, multimodality, long context, and next generation agentic capabilities. ArXiv:2507.06261"},{"key":"4627_CR117","first-page":"26584","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"J Han","year":"2024","unstructured":"Han J, Gong K, Zhang Y, et al. OneLLM: one framework to align all modalities with language. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024. 26584\u201326595"},{"key":"4627_CR118","first-page":"27425","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Z Tang","year":"2024","unstructured":"Tang Z, Yang Z, Khademi M, et al. Codi-2: in-context interleaved and interactive any-to-any generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024. 27425\u201327434"},{"key":"4627_CR119","unstructured":"Team K, Bai Y, Bao Y, et al. KIMI K2: open agentic intelligence. ArXiv:2507.20534"},{"key":"4627_CR120","unstructured":"Zeng A, Lv X, Zheng Q, et al. GLM-4.5: agentic, reasoning, and coding (ARC) foundation models. ArXiv:2508.06471"}],"container-title":["Science China Information Sciences"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-025-4627-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11432-025-4627-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-025-4627-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,30]],"date-time":"2025-10-30T11:03:47Z","timestamp":1761822227000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11432-025-4627-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":120,"journal-issue":{"issue":"11","published-print":{"date-parts":[[2025,11]]}},"alternative-id":["4627"],"URL":"https:\/\/doi.org\/10.1007\/s11432-025-4627-3","relation":{},"ISSN":["1674-733X","1869-1919"],"issn-type":[{"value":"1674-733X","type":"print"},{"value":"1869-1919","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"11 June 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"19 September 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 October 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 October 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"213101"}}