{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,8]],"date-time":"2026-04-08T09:52:43Z","timestamp":1775641963966,"version":"3.50.1"},"reference-count":56,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2026,4,8]],"date-time":"2026-04-08T00:00:00Z","timestamp":1775606400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,4,8]],"date-time":"2026-04-08T00:00:00Z","timestamp":1775606400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"DOI":"10.1007\/s11227-026-08473-x","type":"journal-article","created":{"date-parts":[[2026,4,8]],"date-time":"2026-04-08T09:02:57Z","timestamp":1775638977000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["MGCAF: a multi-granularity cross-modal alignment framework for grounded multimodal named entity recognition"],"prefix":"10.1007","volume":"82","author":[{"given":"Xiaojia","family":"Wu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lingfeng","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lin","family":"Cheng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,4,8]]},"reference":[{"key":"8473_CR1","doi-asserted-by":"crossref","unstructured":"Liu P, Li H, Wang Z, Liu J, Ren Y, Zhu H (2022) Multi-features based semantic augmentation networks for named entity recognition in threat intelligence. In: 2022 26th International Conference on Pattern Recognition (ICPR), pp. 1557\u20131563. IEEE","DOI":"10.1109\/ICPR56361.2022.9956373"},{"key":"8473_CR2","first-page":"8441","volume":"34","author":"Y Luo","year":"2020","unstructured":"Luo Y, Xiao F, Zhao H (2020) Hierarchical contextualized representation for named entity recognition. Proc AAAI ConfArtif Intell 34:8441\u20138448","journal-title":"Proc AAAI ConfArtif Intell"},{"key":"8473_CR3","first-page":"10965","volume":"36","author":"J Li","year":"2022","unstructured":"Li J, Fei H, Liu J, Wu S, Zhang M, Teng C, Ji D, Li F (2022) Unified named entity recognition as word-word relation classification. Proc AAAI Conf Artif Intell 36:10965\u201310973","journal-title":"Proc AAAI Conf Artif Intell"},{"key":"8473_CR4","doi-asserted-by":"crossref","unstructured":"Yan H, Gui T, , J., Guo Q, Zhang Z, Qiu X (2021) A unified generative framework for various ner subtasks. arXiv preprint arXiv:2106.01223","DOI":"10.18653\/v1\/2021.acl-long.451"},{"key":"8473_CR5","doi-asserted-by":"crossref","unstructured":"Li X, Feng J, Meng Y, Han Q, Wu F, Li J (2020) A unified mrc framework for named entity recognition. In: Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics, pp 5849\u20135859 (2020)","DOI":"10.18653\/v1\/2020.acl-main.519"},{"issue":"1","key":"8473_CR6","doi-asserted-by":"publisher","first-page":"50","DOI":"10.1186\/s40537-023-00726-3","volume":"10","author":"N Parveen","year":"2023","unstructured":"Parveen N, Chakrabarti P, Hung BT, Shaik A (2023) Twitter sentiment analysis using gated attention recurrent network. J Big Data 10(1):50","journal-title":"J Big Data"},{"key":"8473_CR7","doi-asserted-by":"crossref","unstructured":"Yu J, Jiang J, Yang L, Xia R (2020) Improving multimodal named entity recognition via entity span detection with unified multimodal transformer. Association for Computational Linguistics","DOI":"10.18653\/v1\/2020.acl-main.306"},{"key":"8473_CR8","doi-asserted-by":"crossref","unstructured":"Xu B, Huang S, Sha C, Wang H (2022) Maf: a general matching and alignment framework for multimodal named entity recognition. In: Proceedings of the Fifteenth ACM International Conference on Web Search and Data Mining, pp. 1215\u20131223","DOI":"10.1145\/3488560.3498475"},{"key":"8473_CR9","doi-asserted-by":"crossref","unstructured":"Li X, Sun G, Liu X (2023) Espvr: entity spans position visual regions for multimodal named entity recognition. In: Findings of the Association for Computational Linguistics: EMNLP 2023, pp. 7785\u20137794","DOI":"10.18653\/v1\/2023.findings-emnlp.522"},{"key":"8473_CR10","doi-asserted-by":"crossref","unstructured":"Wang X, Gui M, Jiang Y, Jia Z, Bach N, Wang T, Huang Z, Huang F, Tu K (2021) Ita: Image-text alignments for multi-modal named entity recognition. arXiv preprint arXiv:2112.06482","DOI":"10.18653\/v1\/2022.naacl-main.232"},{"key":"8473_CR11","doi-asserted-by":"crossref","unstructured":"Bao X, Tian M, Zha Z, Qin B (2023) Mpmrc-mner: A unified mrc framework for multimodal named entity recognition based multimodal prompt. In: Proceedings of the 32nd ACM International Conference on Information and Knowledge Management, pp. 47\u201356","DOI":"10.1145\/3583780.3614975"},{"key":"8473_CR12","first-page":"14347","volume":"35","author":"D Zhang","year":"2021","unstructured":"Zhang D, Wei S, Li S, Wu H, Zhu Q, Zhou G (2021) Multi-modal graph fusion for named entity recognition with targeted visual guidance. Proc AAAI Conf Artif Intell 35:14347\u201314355","journal-title":"Proc AAAI Conf Artif Intell"},{"issue":"16","key":"8473_CR13","doi-asserted-by":"publisher","first-page":"23767","DOI":"10.1007\/s11227-024-06347-8","volume":"80","author":"Y Gong","year":"2024","unstructured":"Gong Y, Lv X, Yuan Z, Wang Z, Hu F, You X (2024) Multimodal heterogeneous graph entity-level fusion for named entity recognition with multi-granularity visual guidance. J Supercomput 80(16):23767\u201323793","journal-title":"J Supercomput"},{"issue":"2","key":"8473_CR14","doi-asserted-by":"publisher","first-page":"715","DOI":"10.1109\/TKDE.2022.3224228","volume":"36","author":"X Zhu","year":"2022","unstructured":"Zhu X, Li Z, Wang X, Jiang X, Sun P, Wang X, Xiao Y, Yuan NJ (2022) Multi-modal knowledge graph construction and application: A survey. IEEE Trans Knowl Data Eng 36(2):715\u2013735","journal-title":"IEEE Trans Knowl Data Eng"},{"key":"8473_CR15","doi-asserted-by":"crossref","unstructured":"Yu J, Li Z, Wang J, Xia R (2023) Grounded multimodal named entity recognition on social media. In: Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 9141\u20139154","DOI":"10.18653\/v1\/2023.acl-long.508"},{"key":"8473_CR16","doi-asserted-by":"crossref","unstructured":"Tang J, Wang Z, Gong Z, Yu J, Zhu X, Yin J (2025) Multi-grained query-guided set prediction network for grounded multimodal named entity recognition. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 39, pp. 25246\u201325254","DOI":"10.1609\/aaai.v39i24.34711"},{"key":"8473_CR17","doi-asserted-by":"crossref","unstructured":"Moon S, Neves L, Carvalho V (2018) Multimodal named entity recognition for short social media posts. arXiv preprint arXiv:1802.07862","DOI":"10.18653\/v1\/N18-1078"},{"key":"8473_CR18","doi-asserted-by":"crossref","unstructured":"Lu D, Neves L, Carvalho V, Zhang N, Ji H (2018) Visual attention model for name tagging in multimodal social media. In: Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 1990\u20131999","DOI":"10.18653\/v1\/P18-1185"},{"key":"8473_CR19","doi-asserted-by":"crossref","unstructured":"Zhang Q, Fu J, Liu X, Huang X (2018) Adaptive co-attention network for named entity recognition in tweets. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 32","DOI":"10.1609\/aaai.v32i1.11962"},{"key":"8473_CR20","doi-asserted-by":"publisher","first-page":"2520","DOI":"10.1109\/TMM.2020.3013398","volume":"23","author":"C Zheng","year":"2020","unstructured":"Zheng C, Wu Z, Wang T, Cai Y, Li Q (2020) Object-aware multimodal named entity recognition in social media posts with adversarial learning. IEEE Trans Multimed 23:2520\u20132532","journal-title":"IEEE Trans Multimed"},{"key":"8473_CR21","doi-asserted-by":"crossref","unstructured":"Chen X, Zhang N, Li L, Yao Y, Deng S, Tan C, Huang F, Si L, Chen H (2022) Good visual guidance makes a better extractor: Hierarchical visual prefix for multimodal entity and relation extraction. arXiv preprint arXiv:2205.03521","DOI":"10.18653\/v1\/2022.findings-naacl.121"},{"key":"8473_CR22","doi-asserted-by":"crossref","unstructured":"Jia M, Shen L, Shen X, Liao L, Chen M, He X, Chen Z, Li J (2023) Mner-qg: An end-to-end mrc framework for multimodal named entity recognition with query grounding. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 37, pp. 8032\u20138040","DOI":"10.1609\/aaai.v37i7.25971"},{"key":"8473_CR23","doi-asserted-by":"crossref","unstructured":"Jia M, Shen X, Shen L, Pang J, Liao L, Song Y, Chen M, He X (2022) Query prior matters: A mrc framework for multimodal named entity recognition. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 3549\u20133558","DOI":"10.1145\/3503161.3548427"},{"key":"8473_CR24","doi-asserted-by":"crossref","unstructured":"Chen X, Zhang N, Li L, Deng S, Tan C, Xu C, Huang F, Si L, Chen H (2022) Hybrid transformer with multi-level fusion for multimodal knowledge graph completion. Proceedings of the 45th International ACM SIGIR Conference on Research and Development in Information Retrieval","DOI":"10.1145\/3477495.3531992"},{"key":"8473_CR25","unstructured":"Wang P, Yang A, Men R, Lin J, Bai S, Li Z, Ma J, Zhou C, Zhou J, Yang H (2022) Ofa: Unifying architectures, tasks, and modalities through a simple sequence-to-sequence learning framework. In: International Conference on Machine Learning, pp. 23318\u201323340. PMLR"},{"key":"8473_CR26","doi-asserted-by":"crossref","unstructured":"Li J, Li H, Sun D, Wang J, Zhang W, Wang Z, Pan G (2024) Llms as bridges: Reformulating grounded multimodal named entity recognition. arXiv preprint arXiv:2402.09989","DOI":"10.18653\/v1\/2024.findings-acl.76"},{"key":"8473_CR27","doi-asserted-by":"crossref","unstructured":"Bao X, Tian M, Wang L, Zha Z, Qin B (2024) Contrastive pre-training with multi-level alignment for grounded multimodal named entity recognition. In: Proceedings of the 2024 International Conference on Multimedia Retrieval, pp. 795\u2013803","DOI":"10.1145\/3652583.3658011"},{"key":"8473_CR28","doi-asserted-by":"publisher","first-page":"1137","DOI":"10.1109\/TPAMI.2016.2577031","volume":"39","author":"S Ren","year":"2015","unstructured":"Ren S, He K, Girshick RB, Sun J (2015) Faster r-cnn: Towards real-time object detection with region proposal networks. IEEE Trans Pattern Anal Mach Intell 39:1137\u20131149","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"8473_CR29","doi-asserted-by":"crossref","unstructured":"Zhang Q, Fu J, Liu X, Huang X (2018) Adaptive co-attention network for named entity recognition in tweets. In: AAAI Conference on Artificial Intelligence","DOI":"10.1609\/aaai.v32i1.11962"},{"key":"8473_CR30","doi-asserted-by":"crossref","unstructured":"Chen L, Ma W, Xiao J, Zhang H, Liu W, Chang S-F (2020) Ref-nms: Breaking proposal bottlenecks in two-stage referring expression grounding. arXiv:2009.01449","DOI":"10.1609\/aaai.v35i2.16188"},{"key":"8473_CR31","doi-asserted-by":"crossref","unstructured":"Yang S, Li G, Yu Y (2020) Graph-structured referring expression reasoning in the wild. 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), 9949\u20139958","DOI":"10.1109\/CVPR42600.2020.00997"},{"key":"8473_CR32","doi-asserted-by":"crossref","unstructured":"Yu L, Lin ZL, Shen X, Yang J, Lu X, Bansal M, Berg TL (2018) Mattnet: Modular attention network for referring expression comprehension. 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 1307\u20131315","DOI":"10.1109\/CVPR.2018.00142"},{"key":"8473_CR33","unstructured":"Redmon J, Farhadi A (2018) Yolov3: An incremental improvement. arXiv:1804.02767"},{"key":"8473_CR34","doi-asserted-by":"crossref","unstructured":"Carion N, Massa F, Synnaeve G, Usunier N, Kirillov A, Zagoruyko S (2020) End-to-end object detection with transformers. arXiv:2005.12872","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"8473_CR35","doi-asserted-by":"crossref","unstructured":"Deng J, Yang Z, Chen T, Zhou W-g, Li H (2021) Transvg: End-to-end visual grounding with transformers. 2021 IEEE\/CVF International Conference on Computer Vision (ICCV), 1749\u20131759","DOI":"10.1109\/ICCV48922.2021.00179"},{"key":"8473_CR36","doi-asserted-by":"crossref","unstructured":"Yang Z, Gong B, Wang L, Huang W, Yu D, Luo J (2019) A fast and accurate one-stage approach to visual grounding. 2019 IEEE\/CVF International Conference on Computer Vision (ICCV), 4682\u20134692","DOI":"10.1109\/ICCV.2019.00478"},{"key":"8473_CR37","doi-asserted-by":"crossref","unstructured":"Ye J, Tian J, Yan M, Yang X, Wang X, Zhang J, He L, Lin X (2022) Shifting more attention to visual backbone: Query-modulated refinement networks for end-to-end visual grounding. 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), 15481\u201315491","DOI":"10.1109\/CVPR52688.2022.01506"},{"key":"8473_CR38","doi-asserted-by":"publisher","first-page":"4334","DOI":"10.1109\/TMM.2023.3321501","volume":"26","author":"L Xiao","year":"2023","unstructured":"Xiao L, Yang X, Peng F, Yan M, Wang Y, Xu C (2023) Clip-vg: Self-paced curriculum adapting of clip for visual grounding. IEEE Trans Multimed 26:4334\u20134347","journal-title":"IEEE Trans Multimed"},{"key":"8473_CR39","doi-asserted-by":"crossref","unstructured":"Xiao L, Yang X, Peng F, Wang Y, Xu C (2024) Hivg: Hierarchical multimodal fine-grained modulation for visual grounding. In: Proceedings of the 32nd ACM International Conference on Multimedia, pp. 5460\u20135469","DOI":"10.1145\/3664647.3681071"},{"key":"8473_CR40","doi-asserted-by":"crossref","unstructured":"Wang Z, Lu Y, Li Q, Tao X, Guo Y, Gong M, Liu T (2022) Cris: Clip-driven referring image segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11686\u201311695","DOI":"10.1109\/CVPR52688.2022.01139"},{"key":"8473_CR41","doi-asserted-by":"crossref","unstructured":"Kim S, Kang M, Kim D, Park J, Kwak S (2024) Extending clip\u2019s image-text alignment to referring image segmentation. In: Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers), pp. 4611\u20134628","DOI":"10.18653\/v1\/2024.naacl-long.258"},{"key":"8473_CR42","unstructured":"Chen K, Zhang Z, Zeng W, Zhang R, Zhu F, Zhao R (2023) Shikra: Unleashing multimodal llm\u2019s referential dialogue magic. arXiv preprint arXiv:2306.15195"},{"key":"8473_CR43","unstructured":"You H, Zhang H, Gan Z, Du X, Zhang B, Wang Z, Cao L, Chang S-F, Yang Y (2024) Ferret: Refer and ground anything anywhere at any granularity. In: The Twelfth International Conference on Learning Representations"},{"key":"8473_CR44","doi-asserted-by":"crossref","unstructured":"Zhang H, Li H, Li F, Ren T, Zou X, Liu S, Huang S, Gao J, Leizhang Li C, et al (2024) Llava-grounding: Grounded visual chat with large multimodal models. In: European Conference on Computer Vision, pp. 19\u201335. Springer","DOI":"10.1007\/978-3-031-72775-7_2"},{"key":"8473_CR45","doi-asserted-by":"crossref","unstructured":"Chen G, Shen L, Shao R, Deng X, Nie L (2024) Lion: Empowering multimodal large language model with dual-level visual knowledge. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 26540\u201326550","DOI":"10.1109\/CVPR52733.2024.02506"},{"key":"8473_CR46","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"8473_CR47","doi-asserted-by":"crossref","unstructured":"Zhang P, Li X, Hu X, Yang J, Zhang L, Wang L, Choi Y, Gao J (2021) Vinvl: Revisiting visual representations in vision-language models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5579\u20135588","DOI":"10.1109\/CVPR46437.2021.00553"},{"key":"8473_CR48","unstructured":"Li J, Li D, Xiong C, Hoi S (2022) Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation. In: International Conference on Machine Learning, pp. 12888\u201312900. PMLR"},{"key":"8473_CR49","doi-asserted-by":"crossref","unstructured":"Smith R (2007) An overview of the tesseract ocr engine. In: Ninth International Conference on Document Analysis and Recognition (ICDAR 2007), vol. 2, pp. 629\u2013633. IEEE","DOI":"10.1109\/ICDAR.2007.4376991"},{"key":"8473_CR50","doi-asserted-by":"crossref","unstructured":"Lewis M, Liu Y, Goyal N, Ghazvininejad M, Mohamed A, Levy O, Stoyanov V, Zettlemoyer L (2019) Bart: Denoising sequence-to-sequence pre-training for natural language generation, translation, and comprehension. arXiv preprint arXiv:1910.13461","DOI":"10.18653\/v1\/2020.acl-main.703"},{"key":"8473_CR51","unstructured":"Chen T, Kornblith S, Norouzi M, Hinton G (2020) A simple framework for contrastive learning of visual representations. In: International Conference on Machine Learning, pp. 1597\u20131607. PmLR"},{"key":"8473_CR52","doi-asserted-by":"crossref","unstructured":"Yu Z, Yu J, Xiang C, Zhao Z, Tian Q, Tao D (2018) Rethinking diversified and discriminative proposal generation for visual grounding. arXiv preprint arXiv:1805.03508","DOI":"10.24963\/ijcai.2018\/155"},{"key":"8473_CR53","doi-asserted-by":"crossref","unstructured":"Szegedy C, Vanhoucke V, Ioffe S, Shlens J, Wojna Z (2016) Rethinking the inception architecture for computer vision. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2818\u20132826","DOI":"10.1109\/CVPR.2016.308"},{"key":"8473_CR54","doi-asserted-by":"crossref","unstructured":"Devlin J, Chang M-W, Lee K, Toutanova K (2019) Bert: Pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (long and Short Papers), pp. 4171\u20134186","DOI":"10.18653\/v1\/N19-1423"},{"key":"8473_CR55","doi-asserted-by":"crossref","unstructured":"Anderson P, He X, Buehler C, Teney D, Johnson M, Gould S, Zhang L (2018) Bottom-up and top-down attention for image captioning and visual question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6077\u20136086","DOI":"10.1109\/CVPR.2018.00636"},{"key":"8473_CR56","unstructured":"Radford A, Kim JW, Hallacy C, Ramesh A, Goh G, Agarwal S, Sastry G, Askell A, Mishkin P, Clark J et al (2021) Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PmLR"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-026-08473-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-026-08473-x","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-026-08473-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,8]],"date-time":"2026-04-08T09:03:25Z","timestamp":1775639005000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-026-08473-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,8]]},"references-count":56,"journal-issue":{"issue":"5","published-online":{"date-parts":[[2026,4]]}},"alternative-id":["8473"],"URL":"https:\/\/doi.org\/10.1007\/s11227-026-08473-x","relation":{},"ISSN":["1573-0484"],"issn-type":[{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,4,8]]},"assertion":[{"value":"29 October 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 March 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 April 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest. The authors have no conflict of interest to declare that are relevant to the content of this article.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"328"}}