{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,19]],"date-time":"2026-01-19T11:05:56Z","timestamp":1768820756879,"version":"3.49.0"},"reference-count":49,"publisher":"Springer Science and Business Media LLC","issue":"8","license":[{"start":{"date-parts":[[2024,1,5]],"date-time":"2024-01-05T00:00:00Z","timestamp":1704412800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,5]],"date-time":"2024-01-05T00:00:00Z","timestamp":1704412800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["72204155"],"award-info":[{"award-number":["72204155"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62202282"],"award-info":[{"award-number":["62202282"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Shanghai Youth Science and Technology Talents Sailing Program","award":["22YF1413700"],"award-info":[{"award-number":["22YF1413700"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Mach Learn"],"published-print":{"date-parts":[[2024,8]]},"DOI":"10.1007\/s10994-023-06456-0","type":"journal-article","created":{"date-parts":[[2024,1,5]],"date-time":"2024-01-05T12:01:58Z","timestamp":1704456118000},"page":"5351-5378","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Entity recognition based on heterogeneous graph reasoning of visual region and text candidate"],"prefix":"10.1007","volume":"113","author":[{"given":"Xinzhi","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6146-9887","authenticated-orcid":false,"given":"Nengjun","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiahao","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yudong","family":"Chang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhennan","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,1,5]]},"reference":[{"key":"6456_CR1","unstructured":"Akbik, A., Blythe, D., & Vollgraf, R. (2018). Contextual string embeddings for sequence labeling. In Proceedings of the 27th international conference on computational linguistics (pp. 1638\u20131649)."},{"key":"6456_CR2","doi-asserted-by":"crossref","unstructured":"Arshad, O., Gallo, I., Nawaz, S., & Calefati, A. (2019). Aiding intra-text representations with visual context for multimodal named entity recognition. In 2019 International conference on document analysis and recognition (ICDAR) (pp. 337\u2013342). IEEE.","DOI":"10.1109\/ICDAR.2019.00061"},{"key":"6456_CR3","doi-asserted-by":"crossref","unstructured":"Asgari-Chenaghlu, M., Feizi-Derakhshi, M. R., Farzinvash, L., Balafar, M., & Motamed, C. (2020). A multimodal deep learning approach for named entity recognition from social media. arXiv preprint arXiv:2001.06888","DOI":"10.1007\/s00521-021-06488-4"},{"key":"6456_CR4","doi-asserted-by":"crossref","unstructured":"Cai, Z., & Vasconcelos, N. (2018). Cascade r-cnn: Delving into high quality object detection. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 6154\u20136162).","DOI":"10.1109\/CVPR.2018.00644"},{"key":"6456_CR5","doi-asserted-by":"crossref","unstructured":"Changpinyo, S., Sharma, P., Ding, N., & Soricut, R. (2021). Conceptual 12m: Pushing web-scale image-text pre-training to recognize long-tail visual concepts. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 3558\u20133568).","DOI":"10.1109\/CVPR46437.2021.00356"},{"key":"6456_CR6","doi-asserted-by":"crossref","unstructured":"Chen, D., Li, Z., Gu, B., & Chen, Z. (2021). Multimodal named entity recognition with image attributes and image knowledge. In Database systems for advanced applications: 26th international conference, DASFAA 2021, Taipei, Taiwan, April 11\u201314, 2021, proceedings, Part II 26 (pp. 186\u2013201). Springer.","DOI":"10.1007\/978-3-030-73197-7_12"},{"key":"6456_CR7","doi-asserted-by":"crossref","unstructured":"Cho, K., Van\u00a0Merri\u00ebnboer, B., Gulcehre, C., Bahdanau, D., Bougares, F., Schwenk, H., & Bengio, Y. (2014). Learning phrase representations using rnn encoder-decoder for statistical machine translation. arXiv preprint arXiv:1406.1078","DOI":"10.3115\/v1\/D14-1179"},{"key":"6456_CR8","unstructured":"Cui, Y., Che, W., Wang, S., & Liu, T. (2022). Lert: A linguistically-motivated pre-trained language model. arXiv preprint arXiv:2211.05344"},{"key":"6456_CR9","unstructured":"Devlin, J., Chang, M. -W., Lee, K., & Toutanova, K. (2018). Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805"},{"key":"6456_CR10","doi-asserted-by":"crossref","unstructured":"Grishman, R., & Sundheim, B. M. (1996). Message understanding conference-6: A brief history. In COLING 1996 volume 1: The 16th international conference on computational linguistics.","DOI":"10.3115\/992628.992709"},{"key":"6456_CR11","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., & Girshick, R. (2017). Mask r-cnn. In Proceedings of the IEEE international conference on computer vision (pp. 2961\u20132969).","DOI":"10.1109\/ICCV.2017.322"},{"key":"6456_CR12","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016). Deep residual learning for image recognition. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 770\u2013778).","DOI":"10.1109\/CVPR.2016.90"},{"key":"6456_CR13","doi-asserted-by":"crossref","unstructured":"Huang, P. -Y., Liu, F., Shiang, S. -R., Oh, J., & Dyer, C. (2016). Attention-based multimodal neural machine translation. In Proceedings of the first conference on machine translation (vol. 2, pp. 639\u2013645).","DOI":"10.18653\/v1\/W16-2360"},{"key":"6456_CR14","unstructured":"Huang, Z., Xu, W., & Yu, K. (2015). Bidirectional lstm-crf models for sequence tagging. arXiv preprint arXiv:1508.01991"},{"key":"6456_CR15","unstructured":"Hudson, D., & Manning, C. D. (2019). Learning by abstraction: The neural state machine. In Advances in neural information processing systems (vol. 32)."},{"key":"6456_CR16","doi-asserted-by":"crossref","unstructured":"Ive, J., Madhyastha, P., & Specia, L. (2019). Distilling translations with visual awareness. arXiv preprint arXiv:1906.07701","DOI":"10.18653\/v1\/P19-1653"},{"key":"6456_CR17","unstructured":"Jiao, Z., Sun, S., & Sun, K. (2018). Chinese lexical analysis with deep bi-gru-crf network. arXiv preprint arXiv:1807.01882"},{"key":"6456_CR18","unstructured":"Kingma, D. P., & Ba, J. (2014). Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980"},{"key":"6456_CR19","unstructured":"Kipf, T. N., & Welling, M. (2016). Semi-supervised classification with graph convolutional networks. arXiv preprint arXiv:1609.02907"},{"key":"6456_CR20","doi-asserted-by":"crossref","unstructured":"Li, Y., Qian, Y., Yu, Y., Qin, X., Zhang, C., Liu, Y., Yao, K., Han, J., Liu, J., & Ding, E. (2021). Structext: Structured text understanding with multi-modal transformers. In Proceedings of the 29th ACM international conference on multimedia (pp. 1912\u20131920).","DOI":"10.1145\/3474085.3475345"},{"key":"6456_CR21","unstructured":"Li, Y., Tarlow, D., Brockschmidt, M., & Zemel, R. (2015). Gated graph sequence neural networks. arXiv preprint arXiv:1511.05493"},{"key":"6456_CR22","doi-asserted-by":"crossref","unstructured":"Lin, H., Meng, F., Su, J., Yin, Y., Yang, Z., Ge, Y., Zhou, J., & Luo, J. (2020). Dynamic context-guided capsule network for multimodal machine translation. In Proceedings of the 28th ACM international conference on multimedia (pp. 1320\u20131329).","DOI":"10.1145\/3394171.3413715"},{"issue":"1","key":"6456_CR23","doi-asserted-by":"publisher","first-page":"50","DOI":"10.1109\/TKDE.2020.2981314","volume":"34","author":"J Li","year":"2020","unstructured":"Li, J., Sun, A., Han, J., & Li, C. (2020). A survey on deep learning for named entity recognition. IEEE Transactions on Knowledge and Data Engineering, 34(1), 50\u201370.","journal-title":"IEEE Transactions on Knowledge and Data Engineering"},{"issue":"4","key":"6456_CR24","doi-asserted-by":"publisher","first-page":"4109","DOI":"10.1007\/s10489-021-02546-5","volume":"52","author":"L Liu","year":"2022","unstructured":"Liu, L., Wang, M., Zhang, M., Qing, L., & He, X. (2022). Uamner: Uncertainty-aware multimodal named entity recognition in social media posts. Applied Intelligence, 52(4), 4109\u20134125.","journal-title":"Applied Intelligence"},{"key":"6456_CR25","doi-asserted-by":"crossref","unstructured":"Lu, D., Neves, L., Carvalho, V., Zhang, N., & Ji, H. (2018). Visual attention model for name tagging in multimodal social media. In Proceedings of the 56th annual meeting of the association for computational linguistics (vol. 1, pp. 1990\u20131999).","DOI":"10.18653\/v1\/P18-1185"},{"key":"6456_CR26","unstructured":"Maas, A. L., Hannun, A. Y., & Ng, A. Y. (2013). Rectifier nonlinearities improve neural network acoustic models. In Proc. Icml (vol. 30, pp. 3). Atlanta, Georgia, USA."},{"key":"6456_CR27","doi-asserted-by":"crossref","unstructured":"Moon, S., Neves, L., & Carvalho, V. (2018). Multimodal named entity recognition for short social media posts. arXiv preprint arXiv:1802.07862","DOI":"10.18653\/v1\/N18-1078"},{"key":"6456_CR28","unstructured":"Radford, A., Kim, J. W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J. (2021). Learning transferable visual models from natural language supervision. In International conference on machine learning (pp. 8748\u20138763). PMLR."},{"key":"6456_CR29","unstructured":"Reimers, N., & Gurevych, I. (2017). Optimal hyperparameters for deep lstm-networks for sequence labeling tasks. arXiv preprint arXiv:1707.06799"},{"key":"6456_CR30","doi-asserted-by":"crossref","unstructured":"Reimers, N., & Gurevych, I. (2020). Making monolingual sentence embeddings multilingual using knowledge distillation. arXiv preprint arXiv:2004.09813","DOI":"10.18653\/v1\/2020.emnlp-main.365"},{"key":"6456_CR31","doi-asserted-by":"crossref","unstructured":"Sennrich, R., Haddow, B., & Birch, A. (2015). Neural machine translation of rare words with subword units. arXiv preprint arXiv:1508.07909","DOI":"10.18653\/v1\/P16-1162"},{"key":"6456_CR32","doi-asserted-by":"crossref","unstructured":"Strubell, E., Verga, P., Belanger, D., & McCallum, A. (2017). Fast and accurate entity recognition with iterated dilated convolutions. arXiv preprint arXiv:1702.02098","DOI":"10.18653\/v1\/D17-1283"},{"key":"6456_CR33","doi-asserted-by":"publisher","first-page":"47","DOI":"10.1016\/j.ins.2020.11.024","volume":"554","author":"J Su","year":"2021","unstructured":"Su, J., Chen, J., Jiang, H., Zhou, C., Lin, H., Ge, Y., Wu, Q., & Lai, Y. (2021). Multi-modal neural machine translation with deep semantic interactions. Information Sciences, 554, 47\u201360.","journal-title":"Information Sciences"},{"key":"6456_CR34","doi-asserted-by":"crossref","unstructured":"Sun, L., Wang, J., Zhang, K., Su, Y., & Weng, F. (2021). Rpbert: A text-image relation propagation-based bert model for multimodal ner. In Proceedings of the AAAI conference on artificial intelligence (vol. 35, pp. 13860\u201313868).","DOI":"10.1609\/aaai.v35i15.17633"},{"key":"6456_CR35","doi-asserted-by":"crossref","unstructured":"Sun, P., Zhang, R., Jiang, Y., Kong, T., Xu, C., Zhan, W., Tomizuka, M., Li, L., Yuan, Z., & Wang, C. (2021). Sparse r-cnn: End-to-end object detection with learnable proposals. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 14454\u201314463).","DOI":"10.1109\/CVPR46437.2021.01422"},{"key":"6456_CR36","doi-asserted-by":"crossref","unstructured":"Tomori, S., Ninomiya, T., & Mori, S. (2016). Domain specific named entity recognition referring to the real world by deep neural networks. In Proceedings of the 54th annual meeting of the association for computational linguistics (vol. 2, pp. 236\u2013242).","DOI":"10.18653\/v1\/P16-2039"},{"key":"6456_CR37","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A. N., Kaiser, \u0141., & Polosukhin, I. (2017). Attention is all you need. In Advances in neural information processing systems (vol. 30)."},{"key":"6456_CR38","unstructured":"Veli\u010dkovi\u0107, P., Cucurull, G., Casanova, A., Romero, A., Lio, P., & Bengio, Y. (2017). Graph attention networks. arXiv preprint arXiv:1710.10903"},{"issue":"20","key":"6456_CR39","first-page":"10","volume":"1050","author":"P Velickovic","year":"2017","unstructured":"Velickovic, P., Cucurull, G., Casanova, A., Romero, A., Lio, P., & Bengio, Y. (2017). Graph attention networks. Stat, 1050(20), 10\u201348550.","journal-title":"Stat"},{"key":"6456_CR40","doi-asserted-by":"crossref","unstructured":"Wang, X., Ye, J., Li, Z., Tian, J., Jiang, Y., Yan, M., Zhang, J., & Xiao, Y. (2022). Cat-mner: Multimodal named entity recognition with knowledge-refined cross-modal attention. In 2022 IEEE international conference on multimedia and expo (ICME) (pp. 1\u20136). IEEE.","DOI":"10.1109\/ICME52920.2022.9859972"},{"key":"6456_CR41","doi-asserted-by":"crossref","unstructured":"Wolf, T., Debut, L., Sanh, V., Chaumond, J., Delangue, C., Moi, A., Cistac, P., Rault, T., Louf, R., & Funtowicz, M. (2020). Transformers: State-of-the-art natural language processing. In Proceedings of the 2020 conference on empirical methods in natural language processing: System demonstrations (pp. 38\u201345).","DOI":"10.18653\/v1\/2020.emnlp-demos.6"},{"key":"6456_CR42","doi-asserted-by":"crossref","unstructured":"Yu, J., Jiang, J., Yang, L., & Xia, R. (2020). Improving multimodal named entity recognition via entity span detection with unified multimodal transformer. In Association for computational linguistics","DOI":"10.18653\/v1\/2020.acl-main.306"},{"key":"6456_CR43","doi-asserted-by":"crossref","unstructured":"Zhai, F., Potdar, S., Xiang, B., & Zhou, B. (2017). Neural models for sequence chunking. In Proceedings of the AAAI conference on artificial intelligence (vol. 31).","DOI":"10.1609\/aaai.v31i1.10995"},{"key":"6456_CR44","unstructured":"Zhang, Z., Chen, K., Wang, R., Utiyama, M., Sumita, E., Li, Z., & Zhao, H. (2020). Neural machine translation with universal visual representation. In International conference on learning representations."},{"key":"6456_CR45","doi-asserted-by":"crossref","unstructured":"Zhang, Q., Fu, J., Liu, X., & Huang, X. (2018). Adaptive co-attention network for named entity recognition in tweets. In Proceedings of the AAAI conference on artificial intelligence (vol. 32).","DOI":"10.1609\/aaai.v32i1.11962"},{"key":"6456_CR46","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Jiang, M., & Zhao, Q. (2021). Explicit knowledge incorporation for visual reasoning. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 1356\u20131365).","DOI":"10.1109\/CVPR46437.2021.00141"},{"key":"6456_CR47","doi-asserted-by":"crossref","unstructured":"Zhang, D., Wei, S., Li, S., Wu, H., Zhu, Q., & Zhou, G. (2021). Multi-modal graph fusion for named entity recognition with targeted visual guidance. In Proceedings of the AAAI conference on artificial intelligence (vol. 35, pp. 14347\u201314355).","DOI":"10.1609\/aaai.v35i16.17687"},{"key":"6456_CR48","doi-asserted-by":"crossref","unstructured":"Zhang, D., Wu, L., Sun, C., Li, S., Zhu, Q., & Zhou, G. (2019). Modeling both context-and speaker-sensitive dependence for emotion detection in multi-speaker conversations. In IJCAI (pp. 5415\u20135421).","DOI":"10.24963\/ijcai.2019\/752"},{"key":"6456_CR49","doi-asserted-by":"publisher","first-page":"2520","DOI":"10.1109\/TMM.2020.3013398","volume":"23","author":"C Zheng","year":"2020","unstructured":"Zheng, C., Wu, Z., Wang, T., Cai, Y., & Li, Q. (2020). Object-aware multimodal named entity recognition in social media posts with adversarial learning. IEEE Transactions on Multimedia, 23, 2520\u20132532.","journal-title":"IEEE Transactions on Multimedia"}],"container-title":["Machine Learning"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-023-06456-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10994-023-06456-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-023-06456-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,27]],"date-time":"2025-11-27T18:04:47Z","timestamp":1764266687000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10994-023-06456-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,1,5]]},"references-count":49,"journal-issue":{"issue":"8","published-print":{"date-parts":[[2024,8]]}},"alternative-id":["6456"],"URL":"https:\/\/doi.org\/10.1007\/s10994-023-06456-0","relation":{},"ISSN":["0885-6125","1573-0565"],"issn-type":[{"value":"0885-6125","type":"print"},{"value":"1573-0565","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,1,5]]},"assertion":[{"value":"19 April 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"31 July 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 October 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 January 2024","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no confict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"This work does not involve any human subjects or animals, so has no ethical concerns.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval"}},{"value":"Not applicable.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to participate"}},{"value":"All authors consent to submission and publication.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}},{"value":"This content has been made available to all.","name":"free","label":"Free to read"}]}}