{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T16:26:37Z","timestamp":1784305597903,"version":"3.55.0"},"reference-count":51,"publisher":"Springer Science and Business Media LLC","issue":"28","license":[{"start":{"date-parts":[[2024,2,8]],"date-time":"2024-02-08T00:00:00Z","timestamp":1707350400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,2,8]],"date-time":"2024-02-08T00:00:00Z","timestamp":1707350400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"DOI":"10.1007\/s11042-024-18472-w","type":"journal-article","created":{"date-parts":[[2024,2,8]],"date-time":"2024-02-08T08:03:57Z","timestamp":1707379437000},"page":"71639-71663","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":8,"title":["MVPN: Multi-granularity visual prompt-guided fusion network for multimodal named entity recognition"],"prefix":"10.1007","volume":"83","author":[{"given":"Wei","family":"Liu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Aiqun","family":"Ren","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4843-1953","authenticated-orcid":false,"given":"Chao","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yan","family":"Peng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shaorong","family":"Xie","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Weimin","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,2,8]]},"reference":[{"key":"18472_CR1","doi-asserted-by":"crossref","unstructured":"Li J, Li H, Pan Z, Pan G (2023) Prompt ChatGPT in MNER: improved multimodal named entity recognition method based on auxiliary refining knowledge from ChatGPT. arXiv:2305.12212","DOI":"10.18653\/v1\/2023.findings-emnlp.184"},{"key":"18472_CR2","unstructured":"Liu P, Li H, Ren Y, Liu J, Si S, Zhu H, Sun L (2023) A novel framework for multimodal named entity recognition with multi-level alignments. arXiv:2305.08372"},{"key":"18472_CR3","doi-asserted-by":"crossref","unstructured":"Cui S, Cao J, Cong X, Sheng J, Li Q, Liu T, Shi J (2023) Enhancing multimodal entity and relation extraction with variational information bottleneck. arXiv:2304.02328","DOI":"10.1109\/TASLP.2023.3345146"},{"key":"18472_CR4","unstructured":"Liu W, Zhong X, Hou J, Li S, Huang H, Fang Y (2023) Integrating large pre-trained models into multimodal named entity recognition with evidential fusion. arXiv:2306.16991"},{"issue":"6","key":"18472_CR5","doi-asserted-by":"publisher","first-page":"2181","DOI":"10.1007\/s13042-022-01754-w","volume":"14","author":"J Chen","year":"2023","unstructured":"Chen J, Xue Y, Zhang H, Ding W, Zhang Z, Chen J (2023) On development of multimodal named entity recognition using part-of-speech and mixture of experts. Int J Mach Learn Cybernet 14(6):2181\u20132192","journal-title":"Int J Mach Learn Cybernet"},{"key":"18472_CR6","doi-asserted-by":"crossref","unstructured":"Wang X, Tian J, Gui M, Li Z, Ye J, Yan M, Xiao Y (2022) PromptMNER: prompt-based entity-related visual clue extraction and integration for multimodal named entity recognition. In: International conference on database systems for advanced applications. Springer, pp 297\u2013305","DOI":"10.1007\/978-3-031-00129-1_24"},{"key":"18472_CR7","doi-asserted-by":"crossref","unstructured":"Liu Y, Li S, Hu F, Liu A, Liu Y (2022) Explicit sparse attention network for multimodal named entity recognition. In: China conference on knowledge graph and semantic computing. Springer, pp 83\u201394","DOI":"10.1007\/978-981-19-7596-7_7"},{"key":"18472_CR8","doi-asserted-by":"crossref","unstructured":"Zhao S, Hu M, Cai Z, Liu F (2021) Modeling dense cross-modal interactions for joint entity-relation extraction. In: Proceedings of the twenty-ninth international conference on international joint conferences on artificial intelligence. pp 4032\u20134038","DOI":"10.24963\/ijcai.2020\/558"},{"key":"18472_CR9","doi-asserted-by":"publisher","unstructured":"Lu D, Neves L, Carvalho V, Zhang N, Ji H (2018) Visual attention model for name tagging in multimodal social media. In: Proceedings of the 56th annual meeting of the association for computational linguistics (vol 1: Long Papers), Association for Computational Linguistics, Melbourne, Australia, pp 1990\u20131999. https:\/\/doi.org\/10.18653\/v1\/P18-1185","DOI":"10.18653\/v1\/P18-1185"},{"key":"18472_CR10","doi-asserted-by":"publisher","unstructured":"Yu J, Jiang J, Yang L, Xia R (2020) Improving multimodal named entity recognition via entity span detection with unified multimodal transformer. In: Proceedings of the 58th annual meeting of the association for computational linguistics, Association for Computational Linguistics, Online, pp 3342\u20133352. https:\/\/doi.org\/10.18653\/v1\/2020.acl-main.306","DOI":"10.18653\/v1\/2020.acl-main.306"},{"issue":"16","key":"18472_CR11","doi-asserted-by":"publisher","first-page":"14347","DOI":"10.1609\/aaai.v35i16.17687","volume":"35","author":"D Zhang","year":"2021","unstructured":"Zhang D, Wei S, Li S, Wu H, Zhu Q, Zhou G (2021) Multi-modal graph fusion for named entity recognition with targeted visual guidance. Proc AAAI Conf Artif Intell 35(16):14347\u201314355. https:\/\/doi.org\/10.1609\/aaai.v35i16.17687","journal-title":"Proc AAAI Conf Artif Intell"},{"key":"18472_CR12","doi-asserted-by":"crossref","unstructured":"Zhang Q, Fu J, Liu X, Huang X (2018) Adaptive co-attention network for named entity recognition in tweets. In: Proceedings of the AAAI conference on artificial intelligence, vol 32","DOI":"10.1609\/aaai.v32i1.11962"},{"key":"18472_CR13","doi-asserted-by":"publisher","first-page":"2520","DOI":"10.1109\/TMM.2020.3013398","volume":"23","author":"C Zheng","year":"2021","unstructured":"Zheng C, Wu Z, Wang T, Cai Y, Li Q (2021) Object-aware multimodal named entity recognition in social media posts with adversarial learning. IEEE Trans Multimed 23:2520\u20132532. https:\/\/doi.org\/10.1109\/TMM.2020.3013398","journal-title":"IEEE Trans Multimed"},{"issue":"2","key":"18472_CR14","doi-asserted-by":"publisher","first-page":"558","DOI":"10.1109\/TKDE.2014.2327042","volume":"27","author":"C Li","year":"2014","unstructured":"Li C, Sun A, Weng J, He Q (2014) Tweet segmentation and its application to named entity recognition. IEEE Trans Knowl Data Eng 27(2):558\u2013570","journal-title":"IEEE Trans Knowl Data Eng"},{"key":"18472_CR15","doi-asserted-by":"crossref","unstructured":"Moon S, Neves L, Carvalho V (2018) Multimodal named entity recognition for short social media posts. In: Proceedings of the 2018 conference of the North American chapter of the association for computational linguistics: human language technologies, vol 1 (Long Papers), pp 852\u2013860","DOI":"10.18653\/v1\/N18-1078"},{"key":"18472_CR16","doi-asserted-by":"crossref","unstructured":"Arshad O, Gallo I, Nawaz S, Calefati A (2019) Aiding intra-text representations with visual context for multimodal named entity recognition. In: 2019 International conference on document analysis and recognition (ICDAR). IEEE, pp 337\u2013342","DOI":"10.1109\/ICDAR.2019.00061"},{"key":"18472_CR17","unstructured":"Liu P, Wang G, Li H, Liu J, Ren Y, Zhu H, Sun L (2022) Multi-granularity cross-modality representation learning for named entity recognition on social media"},{"key":"18472_CR18","doi-asserted-by":"publisher","unstructured":"Wu Z, Zheng C, Cai Y, Chen J, Leung H-f, Li Q (2020) Multimodal representation with embedded visual guiding objects for named entity recognition in social media posts. In: Proceedings of the 28th ACM international conference on multimedia, ACM, Seattle WA USA, pp 1038\u20131046. https:\/\/doi.org\/10.1145\/3394171.3413650","DOI":"10.1145\/3394171.3413650"},{"key":"18472_CR19","first-page":"13860","volume":"35","author":"L Sun","year":"2021","unstructured":"Sun L, Wang J, Zhang K, Su Y, Weng F (2021) RpBERT: a text-image relation propagation-based BERT model for multimodal NER. Proc AAAI Conf Artif Intell 35:13860\u201313868","journal-title":"Proc AAAI Conf Artif Intell"},{"key":"18472_CR20","doi-asserted-by":"publisher","unstructured":"Asgari-Chenaghlu M, Feizi-Derakhshi MR, Farzinvash L, Balafar MA, Motamed C (2022) A multimodal deep learning approach for named entity recognition from social media. Neural Comput Appl 34(3):1905\u20131922. arXiv:2001.06888. https:\/\/doi.org\/10.1007\/s00521-021-06488-4","DOI":"10.1007\/s00521-021-06488-4"},{"key":"18472_CR21","doi-asserted-by":"publisher","first-page":"12","DOI":"10.1016\/j.neucom.2021.01.060","volume":"439","author":"Y Tian","year":"2021","unstructured":"Tian Y, Sun X, Yu H, Li Y, Fu K (2021) Hierarchical self-adaptation network for multimodal named entity recognition in social media. Neurocomputing 439:12\u201321. https:\/\/doi.org\/10.1016\/j.neucom.2021.01.060","journal-title":"Neurocomputing"},{"key":"18472_CR22","doi-asserted-by":"crossref","unstructured":"Xu B, Huang S, Sha C, Wang H (2022) MAF: a general matching and alignment framework for multimodal named entity recognition. In: Proceedings of the fifteenth ACM international conference on web search and data mining, pp 1215\u20131223","DOI":"10.1145\/3488560.3498475"},{"key":"18472_CR23","doi-asserted-by":"crossref","unstructured":"Chen D, Li Z, Gu B, Chen Z (2021) Multimodal named entity recognition with image attributes and image knowledge. In: Database systems for advanced applications: 26th international conference, DASFAA 2021, Taipei, Taiwan, April 11\u201314, 2021, Proceedings, Part II 26, Springer, pp 186\u2013201","DOI":"10.1007\/978-3-030-73197-7_12"},{"key":"18472_CR24","unstructured":"Lu J, Zhang D, Zhang J, Zhang P (2022) Flat multi-modal interaction transformer for named entity recognition. In: Proceedings of the 29th international conference on computational linguistics. pp 2055\u20132064"},{"key":"18472_CR25","doi-asserted-by":"crossref","unstructured":"Wang X, Gui M, Jiang Y, Jia Z, Bach N, Wang T, Huang Z, Tu K (2022) ITA: image-text alignments for multi-modal named entity recognition. In: Proceedings of the 2022 conference of the North American chapter of the association for computational linguistics: human language technologies. pp 3176\u20133189","DOI":"10.18653\/v1\/2022.naacl-main.232"},{"key":"18472_CR26","doi-asserted-by":"crossref","unstructured":"Sang EF, Veenstra J (1999) Representing text chunks. arXiv:9907006","DOI":"10.3115\/977035.977059"},{"key":"18472_CR27","unstructured":"Lafferty J, McCallum A, Pereira FC (2001) Conditional random fields: probabilistic models for segmenting and labeling sequence data"},{"key":"18472_CR28","unstructured":"Kenton JDM-WC, Toutanova LK (2019) Bert: Pre-training of deep bidirectional transformers for language understanding. In: Proceedings of naacL-HLT, vol 1. pp 2"},{"key":"18472_CR29","doi-asserted-by":"crossref","unstructured":"Yang Z, Gong B, Wang L, Huang W, Yu D, Luo J (2019) A fast and accurate one-stage approach to visual grounding. In: Proceedings of the IEEE\/CVF international conference on computer vision. pp 4683\u20134693","DOI":"10.1109\/ICCV.2019.00478"},{"key":"18472_CR30","doi-asserted-by":"crossref","unstructured":"Wang T, Anwer RM, Cholakkal H, Khan FS, Pang Y, Shao L (2019) Learning rich features at high-speed for single-shot object detection. In: Proceedings of the IEEE\/CVF international conference on computer vision. pp 1971\u20131980","DOI":"10.1109\/ICCV.2019.00206"},{"key":"18472_CR31","doi-asserted-by":"crossref","unstructured":"Kim S-W, Kook H-K, Sun J-Y, Kang M-C, Ko S-J (2018) Parallel feature pyramid network for object detection. In: Proceedings of the European conference on computer vision (ECCV). pp 234\u2013250","DOI":"10.1007\/978-3-030-01228-1_15"},{"key":"18472_CR32","unstructured":"Jian S, Kaiming H, Shaoqing R, Xiangyu Z (2016) Deep residual learning for image recognition. In: IEEE Conference on computer vision & pattern recognition. pp 770\u2013778"},{"key":"18472_CR33","doi-asserted-by":"crossref","unstructured":"Chen S, Aguilar G, Neves L, Solorio T (2021) Can images help recognize entities? a study of the role of images for multimodal NER. In: Proceedings of the seventh workshop on noisy user-generated text (W-NUT 2021). pp 87\u201396","DOI":"10.18653\/v1\/2021.wnut-1.11"},{"key":"18472_CR34","unstructured":"Radford A, Kim JW, Hallacy C, Ramesh A, Goh G, Agarwal S, Sastry G, Askell A, Mishkin P, Clark J et al (2021) Learning transferable visual models from natural language supervision. In: International conference on machine learning, PMLR, pp 8748\u20138763"},{"key":"18472_CR35","unstructured":"Maas AL, Hannun AY, Ng AY et al (2013) Rectifier nonlinearities improve neural network acoustic models. In: Proc. Icml, vol 30. Atlanta, GA, pp 3"},{"key":"18472_CR36","doi-asserted-by":"crossref","unstructured":"Wang X, Tian J, Gui M, Li Z, Wang R, Yan M, Chen L, Xiao Y (2022) WikiDiverse: a multimodal entity linking dataset with diversified contextual topics and entity types. arXiv:2204.06347","DOI":"10.18653\/v1\/2022.acl-long.328"},{"key":"18472_CR37","doi-asserted-by":"crossref","unstructured":"Berrar D et al. (2019) Cross-Validation","DOI":"10.1016\/B978-0-12-809633-8.20349-X"},{"key":"18472_CR38","doi-asserted-by":"crossref","unstructured":"Hastie T, Tibshirani R, Friedman JH, Friedman JH (2009) The elements of statistical learning: data mining, inference, and prediction. Springer, vol 2","DOI":"10.1007\/978-0-387-84858-7"},{"key":"18472_CR39","unstructured":"Hart PE, Stork DG, Duda RO (2000) Pattern classification. Wiley Hoboken"},{"key":"18472_CR40","doi-asserted-by":"crossref","unstructured":"Chen X, Zhang N, Li L, Yao Y, Deng S, Tan C, Huang F, Si L, Chen H (2022) Good visual guidance make a better extractor: hierarchical visual prefix for multimodal entity and relation extraction. In: Findings of the association for computational linguistics: NAACL 2022. pp 1607\u20131618","DOI":"10.18653\/v1\/2022.findings-naacl.121"},{"key":"18472_CR41","doi-asserted-by":"crossref","unstructured":"Ma X, Hovy E (2016) End-to-end sequence labeling via bi-directional LSTM-CNNS-CRF. In: Proceedings of the 54th annual meeting of the association for computational linguistics (vol 1: Long Papers), pp 1064\u20131074","DOI":"10.18653\/v1\/P16-1101"},{"key":"18472_CR42","doi-asserted-by":"crossref","unstructured":"Lample G, Ballesteros M, Subramanian S, Kawakami K, Dyer C (2016) Neural architectures for named entity recognition. In: Proceedings of the 2016 conference of the North American chapter of the association for computational linguistics: human language technologies, pp 260\u2013270","DOI":"10.18653\/v1\/N16-1030"},{"key":"18472_CR43","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser \u0141, Polosukhin I (2017) Attention is all you need. Adv Neural Inf Process Syst 30"},{"key":"18472_CR44","unstructured":"Liu P, Wang G, Li H, Liu J, Ren Y, Zhu H, Sun L (2022) Multi-granularity cross-modality representation learning for named entity recognition on social media. arXiv:2210.14163"},{"key":"18472_CR45","doi-asserted-by":"crossref","unstructured":"Wang X, Ye J, Li Z, Tian J, Jiang Y, Yan M, Zhang J, Xiao Y (2022) CAT-MNER: multimodal named entity recognition with knowledge-refined cross-modal attention. In: 2022 IEEE International conference on multimedia and expo (ICME). IEEE, pp 1\u20136","DOI":"10.1109\/ICME52920.2022.9859972"},{"key":"18472_CR46","doi-asserted-by":"crossref","unstructured":"Zhao F, Li C, Wu Z, Xing S, Dai X (2022) Learning from different text-image pairs: a relation-enhanced graph convolutional network for multimodal NER. In: Proceedings of the 30th ACM international conference on multimedia, pp 3983\u20133992","DOI":"10.1145\/3503161.3548228"},{"key":"18472_CR47","doi-asserted-by":"crossref","unstructured":"Zhang X, Yuan J, Li L, Liu J (2023) Reducing the bias of visual objects in multimodal named entity recognition. In: Proceedings of the Sixteenth ACM international conference on web search and data mining, pp 958\u2013966","DOI":"10.1145\/3539597.3570485"},{"key":"18472_CR48","unstructured":"Chen F, Feng Y (2023) Chain-of-thought prompt distillation for multimodal named entity and multimodal relation extraction. arXiv:2306.14122"},{"key":"18472_CR49","doi-asserted-by":"crossref","unstructured":"Wang X, Cai J, Jiang Y, Xie P, Tu K, Lu W (2022) Named entity and relation extraction with multi-modal retrieval. arXiv:2212.01612","DOI":"10.18653\/v1\/2022.findings-emnlp.437"},{"key":"18472_CR50","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"18472_CR51","unstructured":"Loshchilov I, Hutter F (2018) Decoupled weight decay regularization. In: International conference on learning representations"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-024-18472-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-024-18472-w\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-024-18472-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,30]],"date-time":"2024-07-30T17:26:43Z","timestamp":1722360403000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-024-18472-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,2,8]]},"references-count":51,"journal-issue":{"issue":"28","published-online":{"date-parts":[[2024,8]]}},"alternative-id":["18472"],"URL":"https:\/\/doi.org\/10.1007\/s11042-024-18472-w","relation":{},"ISSN":["1573-7721"],"issn-type":[{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,2,8]]},"assertion":[{"value":"21 September 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 January 2024","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 January 2024","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 February 2024","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Not applicable","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical Approval"}},{"value":"The authors declare no competing interests.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}]}}