{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T11:12:15Z","timestamp":1778065935703,"version":"3.51.4"},"publisher-location":"Cham","reference-count":36,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032251558","type":"print"},{"value":"9783032251565","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-25156-5_5","type":"book-chapter","created":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T10:20:38Z","timestamp":1778062838000},"page":"82-101","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Are a\u00a0Thousand Words Better Than a\u00a0Single Picture? Beyond Images - A Framework for\u00a0Multi-modal Knowledge Graph Dataset Enrichment"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5111-4487","authenticated-orcid":false,"given":"Pengyu","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4988-978X","authenticated-orcid":false,"given":"Klim","family":"Zaporojets","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5116-5401","authenticated-orcid":false,"given":"Jie","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7943-2591","authenticated-orcid":false,"given":"Jia-Hong","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0183-6910","authenticated-orcid":false,"given":"Paul","family":"Groth","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,5,7]]},"reference":[{"key":"5_CR1","doi-asserted-by":"publisher","unstructured":"An, B., Chen, B., Han, X., Sun, L.: Accurate text-enhanced knowledge graph representation learning. In: Walker, M., Ji, H., Stent, A. (eds.) Proceedings of the 2018 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long Papers), pp. 745\u2013755. Association for Computational Linguistics, New Orleans, Louisiana (2018). https:\/\/doi.org\/10.18653\/v1\/N18-1068","DOI":"10.18653\/v1\/N18-1068"},{"key":"5_CR2","doi-asserted-by":"publisher","unstructured":"Chen, H., Shen, X., Lv, Q., Wang, J., Ni, X., Ye, J.: SAC-KG: exploiting large language models as skilled automatic constructors for domain knowledge graph. In: Ku, L.W., Martins, A., Srikumar, V. (eds.) Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 4345\u20134360. Association for Computational Linguistics, Bangkok, Thailand (2024). https:\/\/doi.org\/10.18653\/v1\/2024.acl-long.238","DOI":"10.18653\/v1\/2024.acl-long.238"},{"key":"5_CR3","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2025.113355","volume":"316","author":"J Chen","year":"2025","unstructured":"Chen, J., Gao, Y., Ge, M., Li, M.: Ambiguity-aware and high-order relation learning for multi-grained image-text matching. Knowl.-Based Syst. 316, 113355 (2025)","journal-title":"Knowl.-Based Syst."},{"key":"5_CR4","unstructured":"Chen, Z., et al.: Noise-powered multi-modal knowledge graph representation framework. In: Rambow, O., Wanner, L., Apidianaki, M., Al-Khalifa, H., Eugenio, B.D., Schockaert, S. (eds.) Proceedings of the 31st International Conference on Computational Linguistics, pp. 141\u2013155. Association for Computational Linguistics, Abu Dhabi, UAE (2025)"},{"key":"5_CR5","doi-asserted-by":"crossref","unstructured":"Chen, Z., et al.: Knowledge graphs meet multi-modal learning: a comprehensive survey (2024)","DOI":"10.2139\/ssrn.5044404"},{"key":"5_CR6","doi-asserted-by":"publisher","DOI":"10.1016\/j.media.2023.103046","volume":"92","author":"S Dayarathna","year":"2024","unstructured":"Dayarathna, S., Islam, K.T., Uribe, S., Yang, G., Hayat, M., Chen, Z.: Deep learning based synthesis of MRI, CT and PET: review and analysis. Med. Image Anal. 92, 103046 (2024)","journal-title":"Med. Image Anal."},{"key":"5_CR7","doi-asserted-by":"publisher","unstructured":"Firmansyah, A.F., Zahera, H.M., Sherif, M.A., Moussallem, D., Ngomo, A.C.N.: Ants: abstractive entity summarization in knowledge graphs, pp. 133\u2013151. Springer, Heidelberg (2025). https:\/\/doi.org\/10.1007\/978-3-031-94575-5_8","DOI":"10.1007\/978-3-031-94575-5_8"},{"key":"5_CR8","doi-asserted-by":"crossref","unstructured":"Guo, J., et al.: From images to textual prompts: Zero-shot visual question answering with frozen large language models. In: 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 10867\u201310877 (2022)","DOI":"10.1109\/CVPR52729.2023.01046"},{"key":"5_CR9","doi-asserted-by":"crossref","unstructured":"Huang, S., et al.: Temporal graph benchmark for machine learning on temporal graphs. In: Oh, A., Naumann, T., Globerson, A., Saenko, K., Hardt, M., Levine, S. (eds.) Adv. Neural Inf. Process. Syst. 36, 2056\u20132073. Curran Associates, Inc. (2023)","DOI":"10.52202\/075280-0099"},{"key":"5_CR10","doi-asserted-by":"publisher","unstructured":"Klironomos, A., Zhou, B., Zheng, Z., Mohamed, G.E., Paulheim, H., Kharlamov, E.: Realite: enrichment of relation embeddings in knowledge graphs using numeric literals. In: The Semantic Web: 22nd European Semantic Web Conference, ESWC 2025, Portoroz, Slovenia, June 1\u20135, 2025, Proceedings, Part I, pp. 41\u201358. Springer, Cham (2025). https:\/\/doi.org\/10.1007\/978-3-031-94575-5_3","DOI":"10.1007\/978-3-031-94575-5_3"},{"key":"5_CR11","doi-asserted-by":"publisher","unstructured":"Ko, H., Yoo, J., Jeong, O.R.: Mdcke: Multimodal deep-context knowledge extractor that integrates contextual information. Alexandria Eng. J. 119, 478\u2013492 (2025). https:\/\/doi.org\/10.1016\/j.aej.2025.01.119, https:\/\/www.sciencedirect.com\/science\/article\/pii\/S1110016825001474","DOI":"10.1016\/j.aej.2025.01.119"},{"key":"5_CR12","doi-asserted-by":"publisher","first-page":"101","DOI":"10.1007\/978-3-031-78977-9_7","volume-title":"Discovery Science","author":"B Koloski","year":"2025","unstructured":"Koloski, B., Pollak, S., Navigli, R., \u0160krlj, B.: Automl-guided fusion of entity and LLM-based representations for document classification. In: Pedreschi, D., Monreale, A., Guidotti, R., Pellungrini, R., Naretto, F. (eds.) Discovery Science, pp. 101\u2013115. Springer Nature Switzerland, Cham (2025)"},{"key":"5_CR13","doi-asserted-by":"crossref","unstructured":"Kuo, C.W., Kira, Z.: Beyond a pre-trained object detector: cross-modal textual and visual context for image captioning. In: 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 17948\u201317958 (2022)","DOI":"10.1109\/CVPR52688.2022.01744"},{"key":"5_CR14","doi-asserted-by":"publisher","unstructured":"Lee, J., Wang, Y., Li, J., Zhang, M.: Multimodal reasoning with multimodal knowledge graph. In: Ku, L.W., Martins, A., Srikumar, V. (eds.) Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 10767\u201310782. Association for Computational Linguistics, Bangkok, Thailand (2024). https:\/\/doi.org\/10.18653\/v1\/2024.acl-long.579","DOI":"10.18653\/v1\/2024.acl-long.579"},{"key":"5_CR15","unstructured":"Li, J., Li, D., Savarese, S., Hoi, S.: BLIP-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. In: Krause, A., Brunskill, E., Cho, K., Engelhardt, B., Sabato, S., Scarlett, J. (eds.) Proceedings of the 40th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a0202, pp. 19730\u201319742. PMLR (2023)"},{"key":"5_CR16","doi-asserted-by":"crossref","unstructured":"Li, X., Zhao, X., Xu, J., Zhang, Y., Xing, C.: Imf: interactive multimodal fusion model for link prediction. In: Proceedings of the ACM Web Conference 2023, pp. 2572\u20132580 (2023)","DOI":"10.1145\/3543507.3583554"},{"key":"5_CR17","unstructured":"Li, Y., Tian, Y., Huang, Y., Lu, W., Wang, S., Lin, W., Rocha, A.: Fakescope: large multimodal expert model for transparent AI-generated image forensics. arXiv preprint arXiv:2503.24267 (2025)"},{"key":"5_CR18","unstructured":"Lin, B., et al..: Moe-llava: mixture of experts for large vision-language models (2024)"},{"key":"5_CR19","unstructured":"Lin, Z., Zhang, Z., Wang, M., Shi, Y., Wu, X., Zheng, Y.: Multi-modal contrastive representation learning for entity alignment. In: Calzolari, N., et al. (eds.) Proceedings of the 29th International Conference on Computational Linguistics, pp. 2572\u20132584. International Committee on Computational Linguistics, Gyeongju, Republic of Korea (2022)"},{"key":"5_CR20","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"459","DOI":"10.1007\/978-3-030-21348-0_30","volume-title":"The Semantic Web","author":"Y Liu","year":"2019","unstructured":"Liu, Y., Li, H., Garcia-Duran, A., Niepert, M., Onoro-Rubio, D., Rosenblum, D.S.: MMKG: Multi-modal Knowledge Graphs. In: Hitzler, P., et al. (eds.) ESWC 2019. LNCS, vol. 11503, pp. 459\u2013474. Springer, Cham (2019). https:\/\/doi.org\/10.1007\/978-3-030-21348-0_30"},{"key":"5_CR21","doi-asserted-by":"publisher","unstructured":"Lu, X., Wang, L., Jiang, Z., He, S., Liu, S.: Mmkrl: a robust embedding approach for multi-modal knowledge graph representation learning. Appl. Intell. 52(7), 74807497 (2022). https:\/\/doi.org\/10.1007\/s10489-021-02693-9","DOI":"10.1007\/s10489-021-02693-9"},{"key":"5_CR22","doi-asserted-by":"crossref","unstructured":"Mei, K., Talebi, H., Ardakani, M., Patel, V.M., Milanfar, P., Delbracio, M.: The power of context: How multimodality improves image super-resolution. In: Proceedings of the Computer Vision and Pattern Recognition Conference, pp. 23141\u201323152 (2025)","DOI":"10.1109\/CVPR52734.2025.02155"},{"key":"5_CR23","doi-asserted-by":"publisher","unstructured":"Misra, I., Zitnick, C.L., Mitchell, M., Girshick, R.: Seeing through the Human Reporting Bias: Visual Classifiers from Noisy Human-Centric Labels . In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 2930\u20132939. IEEE Computer Society, Los Alamitos, CA, USA (2016). https:\/\/doi.org\/10.1109\/CVPR.2016.320","DOI":"10.1109\/CVPR.2016.320"},{"key":"5_CR24","doi-asserted-by":"publisher","unstructured":"Purohit, D., Chudasama, Y., Rivas, A., Vidal, M.E.: Sparkle: Symbolic capturing of knowledge for knowledge graph enrichment with learning. In: Proceedings of the 12th Knowledge Capture Conference 2023, pp. 44\u201352. K-CAP \u201923, Association for Computing Machinery, New York, NY, USA (2023). https:\/\/doi.org\/10.1145\/3587259.3627547","DOI":"10.1145\/3587259.3627547"},{"key":"5_CR25","doi-asserted-by":"publisher","unstructured":"Rezayi, S., Zhao, H., Kim, S., Rossi, R., Lipka, N., Li, S.: Edge: enriching knowledge graph embeddings with external text. In: Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, pp. 2767\u20132776. Association for Computational Linguistics, Online (2021). https:\/\/doi.org\/10.18653\/v1\/2021.naacl-main.221","DOI":"10.18653\/v1\/2021.naacl-main.221"},{"key":"5_CR26","doi-asserted-by":"crossref","unstructured":"Su, T., Zhang, X., Sheng, J., Zhang, Z., Liu, T.: Loginmea: local-to-global interaction network for multi-modal entity alignment. In: ECAI 2024, pp. 1173\u20131180. IOS Press (2024)","DOI":"10.3233\/FAIA240612"},{"key":"5_CR27","unstructured":"Wang, J., et al.: Git: A generative image-to-text transformer for vision and language (2022)"},{"issue":"04","key":"5_CR28","doi-asserted-by":"publisher","first-page":"6194","DOI":"10.1609\/aaai.v34i04.6085","volume":"34","author":"J Wang","year":"2020","unstructured":"Wang, J., et al.: Logo-2k+: a large-scale logo dataset for scalable logo classification. Proc. AAAI Conf. Artif. Intell. 34(04), 6194\u20136201 (2020). https:\/\/doi.org\/10.1609\/aaai.v34i04.6085","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"5_CR29","doi-asserted-by":"crossref","unstructured":"Wilber, M.J., Fang, C., Jin, H., Hertzmann, A., Collomosse, J., Belongie, S.: Bam! the behance artistic media dataset for recognition beyond photography. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1202\u20131211 (2017)","DOI":"10.1109\/ICCV.2017.136"},{"key":"5_CR30","doi-asserted-by":"publisher","unstructured":"Xiang, Y., Zhang, Z., Chen, J., Chen, X., Lin, Z., Zheng, Y.: OntoEA: ontology-guided entity alignment via joint knowledge graph embedding. In: Zong, C., Xia, F., Li, W., Navigli, R. (eds.) Findings of the Association for Computational Linguistics: ACL-IJCNLP 2021, pp. 1117\u20131128. Association for Computational Linguistics, Online (2021). https:\/\/doi.org\/10.18653\/v1\/2021.findings-acl.96","DOI":"10.18653\/v1\/2021.findings-acl.96"},{"key":"5_CR31","doi-asserted-by":"publisher","unstructured":"Xu, D., Xu, T., Wu, S., Zhou, J., Chen, E.: Relation-enhanced negative sampling for multimodal knowledge graph completion. In: Proceedings of the 30th ACM International Conference on Multimedia. p. 3857\u20133866. MM \u201922, Association for Computing Machinery, New York, NY, USA (2022). https:\/\/doi.org\/10.1145\/3503161.3548388","DOI":"10.1145\/3503161.3548388"},{"key":"5_CR32","doi-asserted-by":"publisher","unstructured":"Zhang, Y., et al.: Native: Multi-modal knowledge graph completion in the wild. In: Proceedings of the 47th International ACM SIGIR Conference on Research and Development in Information Retrieval, pp. 91\u2013101. SIGIR \u201924, Association for Computing Machinery, New York, NY, USA (2024). https:\/\/doi.org\/10.1145\/3626772.3657800","DOI":"10.1145\/3626772.3657800"},{"key":"5_CR33","doi-asserted-by":"publisher","unstructured":"Zhang, Y., Chen, Z., Guo, L., Xu, Y., Hu, B., Liu, Z., Zhang, W., Chen, H.: Tokenization, fusion, and augmentation: towards fine-grained multi-modal entity representation. In: Proceedings of the Thirty-Ninth AAAI Conference on Artificial Intelligence and Thirty-Seventh Conference on Innovative Applications of Artificial Intelligence and Fifteenth Symposium on Educational Advances in Artificial Intelligence. AAAI\u201925\/IAAI\u201925\/EAAI\u201925, AAAI Press (2025). https:\/\/doi.org\/10.1609\/aaai.v39i12.33454","DOI":"10.1609\/aaai.v39i12.33454"},{"key":"5_CR34","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Chen, Z., Liang, L., Chen, H., Zhang, W.: Unleashing the power of imbalanced modality information for multi-modal knowledge graph completion. In: Calzolari, N., Kan, M.Y., Hoste, V., Lenci, A., Sakti, S., Xue, N. (eds.) Proceedings of the 2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation (LREC-COLING 2024), pp. 17120\u201317130. ELRA and ICCL, Torino, Italia (2024)","DOI":"10.63317\/2yy5xbfapbho"},{"key":"5_CR35","doi-asserted-by":"publisher","unstructured":"Zhao, W., Wu, X.: Boosting entity-aware image captioning with multi-modal knowledge graph. Trans. Multi. 26, 26592670 (2024). https:\/\/doi.org\/10.1109\/TMM.2023.3301279","DOI":"10.1109\/TMM.2023.3301279"},{"key":"5_CR36","doi-asserted-by":"publisher","unstructured":"Zhao, Y., et al.: MoSE: Modality split and ensemble for multimodal knowledge graph completion. In: Goldberg, Y., Kozareva, Z., Zhang, Y. (eds.) Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing, pp. 10527\u201310536. Association for Computational Linguistics, Abu Dhabi, United Arab Emirates (2022). https:\/\/doi.org\/10.18653\/v1\/2022.emnlp-main.719, https:\/\/aclanthology.org\/2022.emnlp-main.719\/","DOI":"10.18653\/v1\/2022.emnlp-main.719"}],"container-title":["Lecture Notes in Computer Science","The Semantic Web"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-25156-5_5","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T10:21:00Z","timestamp":1778062860000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-25156-5_5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9783032251558","9783032251565"],"references-count":36,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-25156-5_5","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"7 May 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors declare that they have no competing interests relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"ESWC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Semantic Web Conference","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Dubrovnik","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Croatia","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"10 May 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"14 May 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"esws2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/2026.eswc-conferences.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}