{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,8]],"date-time":"2026-04-08T20:38:23Z","timestamp":1775680703939,"version":"3.50.1"},"reference-count":46,"publisher":"Springer Science and Business Media LLC","issue":"16","license":[{"start":{"date-parts":[[2024,7,22]],"date-time":"2024-07-22T00:00:00Z","timestamp":1721606400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,7,22]],"date-time":"2024-07-22T00:00:00Z","timestamp":1721606400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"the National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["61663041"],"award-info":[{"award-number":["61663041"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"the National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62171043"],"award-info":[{"award-number":["62171043"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"the Natural Science Foundation of Qinghai Province","award":["2023-ZJ-916M"],"award-info":[{"award-number":["2023-ZJ-916M"]}]},{"name":"the Beijing Advanced Innovation Center for Future Blockchain and Privacy Computing, Central Leading Local Project \"Fujian Mental Health Human-Computer Interaction Technology Research Center\"","award":["2020L3024"],"award-info":[{"award-number":["2020L3024"]}]},{"name":"the R&D Program of Beijing Municipal Education Commission","award":["KM202111232001"],"award-info":[{"award-number":["KM202111232001"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2024,11]]},"DOI":"10.1007\/s11227-024-06347-8","type":"journal-article","created":{"date-parts":[[2024,7,22]],"date-time":"2024-07-22T16:02:02Z","timestamp":1721664122000},"page":"23767-23793","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Multimodal heterogeneous graph entity-level fusion for named entity recognition with multi-granularity visual guidance"],"prefix":"10.1007","volume":"80","author":[{"given":"Yunchao","family":"Gong","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xueqiang","family":"Lv","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhu","family":"Yuan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"ZhaoJun","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Feng","family":"Hu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xindong","family":"You","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,7,22]]},"reference":[{"key":"6347_CR1","doi-asserted-by":"publisher","first-page":"101958","DOI":"10.1016\/j.inffus.2023.101958","volume":"100","author":"C Zhu","year":"2023","unstructured":"Zhu C, Chen M, Zhang S, Sun C, Liang H, Liu Y, Chen J (2023) SKEAFN: sentiment knowledge enhanced attention fusion network for multimodal sentiment analysis. Info Fusion 100:101958","journal-title":"Info Fusion"},{"key":"6347_CR2","doi-asserted-by":"crossref","unstructured":"Yuan L, Cai Y, Wang J, Li Q (2023) Joint multimodal entity-relation extraction based on edge-enhanced graph alignment network and word-pair relation tagging. In: Proceedings of the AAAI Conference on Artificial Intelligence 37: 11051\u201311059","DOI":"10.1609\/aaai.v37i9.26309"},{"key":"6347_CR3","doi-asserted-by":"publisher","DOI":"10.1109\/TAI.2024.3354668","author":"T Tayir","year":"2024","unstructured":"Tayir T, Li L, Li B, Liu J, Lee KA (2024) Encoder-decoder calibration for multimodal machine translation. IEEE Trans Artif Intell. https:\/\/doi.org\/10.1109\/TAI.2024.3354668","journal-title":"IEEE Trans Artif Intell"},{"key":"6347_CR4","doi-asserted-by":"crossref","unstructured":"Zhang Q, Fu J, Liu X, Huang X (2018) Adaptive co-attention network for named entity recognition in tweets. In: Proceedings of the AAAI Conference on Artificial Intelligence 32(1)","DOI":"10.1609\/aaai.v32i1.11962"},{"key":"6347_CR5","doi-asserted-by":"crossref","unstructured":"Moon S, Neves L, Carvalho V (2018) Multimodal named entity recognition for short social media posts. In: Proceedings of the 2018 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, pp. 852\u2013 860","DOI":"10.18653\/v1\/N18-1078"},{"key":"6347_CR6","doi-asserted-by":"crossref","unstructured":"Lu D, Neves L, Carvalho V, Zhang N, Ji H (2018) Visual attention model for name tagging in multimodal social media. In: Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics, pp. 1990\u2013 1999","DOI":"10.18653\/v1\/P18-1185"},{"issue":"3","key":"6347_CR7","doi-asserted-by":"publisher","first-page":"1905","DOI":"10.1007\/s00521-021-06488-4","volume":"34","author":"M Asgari-Chenaghlu","year":"2021","unstructured":"Asgari-Chenaghlu M, Feizi-Derakhshi MR, Farzinvash L, Balafar MA, Motamed C (2021) CWI: a multimodal deep learning approach for named entity recognition from social media using character, word and image features. Neural Comput Appl 34(3):1905\u20131922","journal-title":"Neural Comput Appl"},{"key":"6347_CR8","doi-asserted-by":"publisher","first-page":"12","DOI":"10.1016\/j.neucom.2021.01.060","volume":"439","author":"Y Tian","year":"2021","unstructured":"Tian Y, Sun X, Yu H, Li Y, Fu K (2021) Hierarchical self-adaptation network for multimodal named entity recognition in social media. Neurocomputing 439:12\u201321","journal-title":"Neurocomputing"},{"key":"6347_CR9","doi-asserted-by":"crossref","unstructured":"Xu B, Huang S, Sha C, Wang H (2022) Maf: a general matching and alignment framework for multimodal named entity recognition. In: Proceedings of the Fifteenth ACM International Conference on Web Search and Data Mining, pp. 1215\u2013 1223","DOI":"10.1145\/3488560.3498475"},{"key":"6347_CR10","doi-asserted-by":"crossref","unstructured":"Wang X, Gui M, Jiang Y, Jia Z, Bach N, Wang T, Huang Z, Tu K (2022) ITA: image-text alignments for multi-modal named entity recognition. In: Proceedings of the 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, pp. 3176\u2013 3189","DOI":"10.18653\/v1\/2022.naacl-main.232"},{"issue":"6","key":"6347_CR11","first-page":"1","volume":"50","author":"H Wang","year":"2024","unstructured":"Wang H, Xu X, Tong W, Chen F (2024) Multi-scale visual semantic enhancement for multi-modal named entity recognition method. Acta Autom Sinica 50(6):1\u201312","journal-title":"Acta Autom Sinica"},{"issue":"4","key":"6347_CR12","doi-asserted-by":"publisher","first-page":"4109","DOI":"10.1007\/s10489-021-02546-5","volume":"52","author":"L Liu","year":"2021","unstructured":"Liu L, Wang M, Zhang M, Qing L, He X (2021) UAMNer: uncertainty-aware multimodal named entity recognition in social media posts. Appl Intell 52(4):4109\u20134125","journal-title":"Appl Intell"},{"key":"6347_CR13","doi-asserted-by":"crossref","unstructured":"Yu J, Jiang J, Yang L, Xia R (2020) Improving multimodal named entity recognition via entity span detection with unified multimodal transformer. In: Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics, pp. 3342\u2013 3352","DOI":"10.18653\/v1\/2020.acl-main.306"},{"key":"6347_CR14","doi-asserted-by":"crossref","unstructured":"Zhang D, Wei S, Li S, Wu H, Zhu Q, Zhou G (2021) Multi-modal graph fusion for named entity recognition with targeted visual guidance. In: Proceedings of the AAAI Conference on Artificial Intelligence 35(16):14347\u201314355","DOI":"10.1609\/aaai.v35i16.17687"},{"key":"6347_CR15","doi-asserted-by":"crossref","unstructured":"Wang X Ye J Li Z, Tian J, Jiang Y, Yan M, Zhang J, Xiao Y (2022) CAT-MNER: multimodal named entity recognition with knowledge-refined cross-modal attention. In: 2022 IEEE International Conference on Multimedia and Expo (ICME), pp. 1\u2013 6","DOI":"10.1109\/ICME52920.2022.9859972"},{"key":"6347_CR16","doi-asserted-by":"crossref","unstructured":"Jia M, Shen X, Shen L, Pang J, Liao L, Song Y, Chen M, He X (2022) Query prior matters: a MRC framework for multimodal named entity recognition. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 3549\u2013 3558","DOI":"10.1145\/3503161.3548427"},{"key":"6347_CR17","doi-asserted-by":"publisher","first-page":"2520","DOI":"10.1109\/TMM.2020.3013398","volume":"23","author":"C Zheng","year":"2021","unstructured":"Zheng C, Wu Z, Wang T, Cai Y, Li Q (2021) Object-aware multimodal named entity recognition in social media posts with adversarial learning. IEEE Trans Multimed 23:2520\u20132532","journal-title":"IEEE Trans Multimed"},{"issue":"6","key":"6347_CR18","doi-asserted-by":"publisher","first-page":"2181","DOI":"10.1007\/s13042-022-01754-w","volume":"14","author":"J Chen","year":"2022","unstructured":"Chen J, Xue Y, Zhang H, Ding W, Zhang Z, Chen J (2022) On development of multimodal named entity recognition using part-of-speech and mixture of experts. Int J Mach Learn Cybern 14(6):2181\u20132192","journal-title":"Int J Mach Learn Cybern"},{"key":"6347_CR19","doi-asserted-by":"crossref","unstructured":"Chen X, Zhang N, Li L, Yao Y, Deng S, Tan C, Huang F, Si L, Chen H (2022) Good visual guidance make a better extractor: Hierarchical visual prefix for multimodal entity and relation extraction. arXiv:2205.03521","DOI":"10.18653\/v1\/2022.findings-naacl.121"},{"key":"6347_CR20","doi-asserted-by":"crossref","unstructured":"Zhao F, Li C, Wu Z, Xing S, Dai X (2022) Learning from different text-image pairs: a relation-enhanced graph convolutional network for multimodal NER. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 3983\u2013 3992","DOI":"10.1145\/3503161.3548228"},{"issue":"10","key":"6347_CR21","doi-asserted-by":"publisher","first-page":"4411","DOI":"10.1007\/s10115-023-01908-4","volume":"65","author":"Y Ren","year":"2023","unstructured":"Ren Y, Li H, Liu P, Liu J, Li Z, Zhu H, Sun L (2023) Owner name entity recognition in websites based on heterogeneous and dynamic graph transformer. Knowl Info Syst 65(10):4411\u20134429","journal-title":"Knowl Info Syst"},{"key":"6347_CR22","doi-asserted-by":"crossref","unstructured":"Jiang B, Zhang Z, Lin D, Tang J, Luo B (2019) Semi-supervised learning with graph learning-convolutional networks. In: 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 11305\u2013 11312","DOI":"10.1109\/CVPR.2019.01157"},{"key":"6347_CR23","unstructured":"Velickovic P, Cucurull G, Casanova A, Romero A, Lio\u2019 P, Bengio Y (2017) Graph attention networks. arXiv:1710.10903"},{"key":"6347_CR24","doi-asserted-by":"crossref","unstructured":"Ishiwatari T, Yasuda Y, Miyazaki T, Goto J (2020) Relation-aware graph attention networks with relational position encodings for emotion recognition in conversations. In: Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP), pp. 7360\u2013 7370","DOI":"10.18653\/v1\/2020.emnlp-main.597"},{"key":"6347_CR25","doi-asserted-by":"crossref","unstructured":"Linmei H, Yang T, Shi C, Ji H, Li X (2019) Heterogeneous graph attention networks for semi-supervised short text classification. In: Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP), pp. 4821\u2013 4830","DOI":"10.18653\/v1\/D19-1488"},{"key":"6347_CR26","doi-asserted-by":"crossref","unstructured":"Hu Z, Dong Y, Wang K, Sun Y (2020) Heterogeneous graph transformer. In: Proceedings of The Web Conference 2020, pp. 2704\u2013 2710","DOI":"10.1145\/3366423.3380027"},{"key":"6347_CR27","doi-asserted-by":"publisher","first-page":"1680","DOI":"10.1109\/LSP.2020.3025128","volume":"27","author":"S Chen","year":"2020","unstructured":"Chen S, Li Z, Tang Z (2020) Relation R-CNN: a graph based relation-aware network for object detection. IEEE Signal Process Lett 27:1680\u20131684","journal-title":"IEEE Signal Process Lett"},{"key":"6347_CR28","doi-asserted-by":"publisher","first-page":"103923","DOI":"10.1016\/j.jvcir.2023.103923","volume":"96","author":"S Chen","year":"2023","unstructured":"Chen S, Yang X, Li Z (2023) Improving semantic segmentation with knowledge reasoning network. J Vis Commun Image Represent 96:103923","journal-title":"J Vis Commun Image Represent"},{"issue":"1","key":"6347_CR29","doi-asserted-by":"publisher","first-page":"103538","DOI":"10.1016\/j.ipm.2023.103538","volume":"61","author":"Q Lu","year":"2024","unstructured":"Lu Q, Sun X, Gao Z, Long Y, Feng J, Zhang H (2024) Coordinated-joint translation fusion framework with sentiment-interactive graph convolutional networks for multimodal sentiment analysis. Info Process Manage 61(1):103538","journal-title":"Info Process Manage"},{"key":"6347_CR30","doi-asserted-by":"publisher","first-page":"127112","DOI":"10.1016\/j.neucom.2023.127112","volume":"569","author":"F Xu","year":"2024","unstructured":"Xu F, Zeng L, Huang Q, Yan K, Wang M, Sheng VS (2024) Hierarchical graph attention networks for multi-modal rumor detection on social media. Neurocomputing 569:127112","journal-title":"Neurocomputing"},{"key":"6347_CR31","doi-asserted-by":"publisher","first-page":"111251","DOI":"10.1016\/j.knosys.2023.111251","volume":"286","author":"K Sun","year":"2024","unstructured":"Sun K, Xie Z, Guo C, Zhang H, Li Y (2024) SDGIN: structure-aware dual-level graph interactive network with semantic roles for visual dialog. Knowl-Based Syst 286:111251","journal-title":"Knowl-Based Syst"},{"key":"6347_CR32","doi-asserted-by":"publisher","first-page":"6345","DOI":"10.18653\/v1\/2022.findings-emnlp.473","volume-title":"Findings of the association for computational linguistics: EMNLP 2022","author":"G Zhao","year":"2022","unstructured":"Zhao G, Dong G, Shi Y, Yan H, Xu W, Li S (2022) Entity-level interaction via heterogeneous graph for multimodal named entity recognition. In: Yoav G, Zornitsa K, Yue Z (eds) Findings of the association for computational linguistics: EMNLP 2022. Association for Computational Linguistics, Baltimore, pp 6345\u20136350"},{"key":"6347_CR33","doi-asserted-by":"crossref","unstructured":"Sang EFTK, Veenstra J (1999) Representing text chunks. In: Ninth Conference of the European Chapter of the Association for Computational Linguistics, pp. 173\u2013 179","DOI":"10.3115\/977035.977059"},{"key":"6347_CR34","unstructured":"Devlin J, Chang M-W, Lee K, Toutanova K (2019) BERT: pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, pp. 4171\u2013 4186"},{"key":"6347_CR35","doi-asserted-by":"crossref","unstructured":"Szegedy C, Vanhoucke V, Ioffe S, Shlens J, Wojna Z (2016) Rethinking the inception architecture for computer vision. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 2818\u2013 2826","DOI":"10.1109\/CVPR.2016.308"},{"key":"6347_CR36","doi-asserted-by":"crossref","unstructured":"He K, Gkioxari G, Doll\u00e1r P, Girshick R (2017) Mask R-CNN. In: 2017 IEEE International Conference on Computer Vision (ICCV), pp. 2980\u2013 2988","DOI":"10.1109\/ICCV.2017.322"},{"key":"6347_CR37","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser L, Polosukhin I (2017) Attention is all you need. In: Proceedings of the 31st International Conference on Neural Information Processing Systems, pp. 6000\u2013 6010"},{"key":"6347_CR38","doi-asserted-by":"crossref","unstructured":"Lison P, Barnes J, Hubin A, Touileb S (2020) Named entity recognition without labelled data: a weak supervision approach. In: Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics, pp. 1518\u2013 1533","DOI":"10.18653\/v1\/2020.acl-main.139"},{"key":"6347_CR39","doi-asserted-by":"crossref","unstructured":"Yang Z, Gong B, Wang L, Huang W, Yu D, Luo J (2019) A fast and accurate one-stage approach to visual grounding. In: 2019 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 4682\u2013 4692","DOI":"10.1109\/ICCV.2019.00478"},{"key":"6347_CR40","unstructured":"Huang Z, Xu W, Yu K (2015) Bidirectional LSTM-CRF models for sequence tagging. arXiv:1508.01991"},{"key":"6347_CR41","doi-asserted-by":"crossref","unstructured":"Lample G, Ballesteros M, Subramanian S, Kawakami K, Dyer C (2016) Neural architectures for named entity recognition. In: Proceedings of the 2016 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, pp. 260\u2013 270","DOI":"10.18653\/v1\/N16-1030"},{"key":"6347_CR42","doi-asserted-by":"crossref","unstructured":"Wu Z, Zheng C, Cai Y, Chen J, Leung H-F, Li Q (2020) Multimodal representation with embedded visual guiding objects for named entity recognition in social media posts. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 1038\u2013 1046","DOI":"10.1145\/3394171.3413650"},{"key":"6347_CR43","doi-asserted-by":"crossref","unstructured":"Chen D, Li Z, Gu B, Chen Z (2021) Multimodal named entity recognition with image attributes and image knowledge. In: Database Systems for Advanced Applications: 26th International Conference, DASFAA 2021, pp. 186\u2013 201","DOI":"10.1007\/978-3-030-73197-7_12"},{"key":"6347_CR44","unstructured":"Liu P, Wang G-S, Li H, Liu J, Ren Y, Zhu H, Sun L (2020) Multi-granularity cross-modality representation learning for named entity recognition on social media. arXiv:2210.14163"},{"key":"6347_CR45","doi-asserted-by":"crossref","unstructured":"Jia M, Shen L, Shen X, Liao L, Chen M, He X, Chen Z, Li J (2023) MNER-QG: an end-to-end MRC framework for multimodal named entity recognition with query grounding. In: Proceedings of the AAAI Conference on Artificial Intelligence 37(7):8032\u20138040","DOI":"10.1609\/aaai.v37i7.25971"},{"key":"6347_CR46","doi-asserted-by":"crossref","unstructured":"Zhang X, Yuan J, Li L, Liu J (2023) Reducing the bias of visual objects in multimodal named entity recognition. In: Proceedings of the Sixteenth ACM International Conference on Web Search and Data Mining, pp. 958\u2013 966","DOI":"10.1145\/3539597.3570485"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-024-06347-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-024-06347-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-024-06347-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,8,24]],"date-time":"2024-08-24T12:18:39Z","timestamp":1724501919000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-024-06347-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,7,22]]},"references-count":46,"journal-issue":{"issue":"16","published-print":{"date-parts":[[2024,11]]}},"alternative-id":["6347"],"URL":"https:\/\/doi.org\/10.1007\/s11227-024-06347-8","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"value":"0920-8542","type":"print"},{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,7,22]]},"assertion":[{"value":"5 July 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 July 2024","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no conflict of interest to declare that are relevant to the content of this article.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}