{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T14:16:09Z","timestamp":1783692969929,"version":"3.55.0"},"reference-count":50,"publisher":"Elsevier BV","issue":"1","license":[{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Information Processing &amp; Management"],"published-print":{"date-parts":[[2027,1]]},"DOI":"10.1016\/j.ipm.2026.105041","type":"journal-article","created":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T10:54:34Z","timestamp":1783680874000},"page":"105041","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PA","title":["Dual-Tower Multimodal Entity Linking with Deep Interaction"],"prefix":"10.1016","volume":"64","author":[{"given":"Huayu","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-6109-0132","authenticated-orcid":false,"given":"Xiaotong","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qi","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chenli","family":"Shen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiang","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.ipm.2026.105041_b1","series-title":"European conference on information retrieval","first-page":"463","article-title":"Multimodal entity linking for tweets","author":"Adjali","year":"2020"},{"key":"10.1016\/j.ipm.2026.105041_b2","series-title":"Proceedings of the 2022 conference of the North American chapter of the association for computational linguistics: human language technologies: industry track","first-page":"209","article-title":"RefinED: An efficient zero-shot-capable approach to end-to-end entity linking","author":"Ayoola","year":"2022"},{"key":"10.1016\/j.ipm.2026.105041_b3","series-title":"Knowledge graphs meet multi-modal learning: A comprehensive survey","author":"Chen","year":"2024"},{"key":"10.1016\/j.ipm.2026.105041_b4","series-title":"Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: human language technologies, volume 1 (long and short papers)","first-page":"4171","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2019"},{"key":"10.1016\/j.ipm.2026.105041_b5","doi-asserted-by":"crossref","unstructured":"Dou, Z.-Y., Xu, Y., Gan, Z., Wang, J., Wang, S., Wang, L., Zhu, C., Zhang, P., Yuan, L., Peng, N., et al. (2022). An empirical study of training end-to-end vision-and-language transformers. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 18166\u201318176).","DOI":"10.1109\/CVPR52688.2022.01763"},{"key":"10.1016\/j.ipm.2026.105041_b6","doi-asserted-by":"crossref","unstructured":"Hu, Z., Guti\u00e9rrez-Basulto, V., Li, R., & Pan, J. Z. (2025). Multi-level matching network for multimodal entity linking. In Proceedings of the 31st ACM SIGKDD conference on knowledge discovery and data mining v. 1 (pp. 508\u2013519).","DOI":"10.1145\/3690624.3709306"},{"key":"10.1016\/j.ipm.2026.105041_b7","doi-asserted-by":"crossref","unstructured":"Hu, Z., Guti\u00e9rrez-Basulto, V., Xiang, Z., Li, R., & Pan, J. Z. (2025). Multi-level mixture of experts for multimodal entity linking. In Proceedings of the 31st ACM SIGKDD conference on knowledge discovery and data mining v. 2 (pp. 979\u2013990).","DOI":"10.1145\/3711896.3737060"},{"key":"10.1016\/j.ipm.2026.105041_b8","series-title":"Natural language processing and Chinese computing","first-page":"343","article-title":"A multilevel interaction network framework for multimodal entity linking","author":"Jia","year":"2025"},{"issue":"1","key":"10.1016\/j.ipm.2026.105041_b9","doi-asserted-by":"crossref","DOI":"10.1016\/j.ipm.2022.103120","article-title":"Multimodal fake news detection via progressive fusion networks","volume":"60","author":"Jing","year":"2023","journal-title":"Information Processing & Management"},{"key":"10.1016\/j.ipm.2026.105041_b10","doi-asserted-by":"crossref","unstructured":"Kim, J., Lee, G., Kim, T., & Shin, K. (2025). KGMEL: Knowledge Graph-Enhanced Multimodal Entity Linking. In Proceedings of the 48th international ACM SIGIR conference on research and development in information retrieval (pp. 3015\u20133019).","DOI":"10.1145\/3726302.3730217"},{"key":"10.1016\/j.ipm.2026.105041_b11","series-title":"International conference on machine learning","first-page":"5583","article-title":"Vilt: Vision-and-language transformer without convolution or region supervision","author":"Kim","year":"2021"},{"key":"10.1016\/j.ipm.2026.105041_b12","first-page":"1","article-title":"Advances and challenges in multimodal entity linking: A comprehensive survey","author":"Li","year":"2025","journal-title":"International Journal of Software Engineering and Knowledge Engineering"},{"key":"10.1016\/j.ipm.2026.105041_b13","first-page":"9694","article-title":"Align before fuse: Vision and language representation learning with momentum distillation","volume":"34","author":"Li","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"4","key":"10.1016\/j.ipm.2026.105041_b14","doi-asserted-by":"crossref","DOI":"10.1016\/j.ipm.2023.103348","article-title":"Knowledge graph representation learning with simplifying hierarchical feature propagation","volume":"60","author":"Li","year":"2023","journal-title":"Information Processing & Management"},{"issue":"11","key":"10.1016\/j.ipm.2026.105041_b15","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3656579","article-title":"A survey of multi-modal knowledge graphs: Technologies and trends","volume":"56","author":"Liang","year":"2024","journal-title":"ACM Computing Surveys"},{"key":"10.1016\/j.ipm.2026.105041_b16","series-title":"Unimel: A unified framework for multimodal entity linking with large language models","author":"Liu","year":"2024"},{"key":"10.1016\/j.ipm.2026.105041_b17","series-title":"Visual instruction tuning","author":"Liu","year":"2023"},{"key":"10.1016\/j.ipm.2026.105041_b18","doi-asserted-by":"crossref","unstructured":"Liu, Z., Li, J., Li, K., Ruan, T., Wang, C., He, X., Wang, Z., Cao, X., & Liu, J. (2025). I2CR: Intra-and Inter-modal Collaborative Reflections for Multimodal Entity Linking. In Proceedings of the 33rd ACM international conference on multimedia (pp. 4942\u20134951).","DOI":"10.1145\/3746027.3755674"},{"key":"10.1016\/j.ipm.2026.105041_b19","series-title":"Roberta: A robustly optimized bert pretraining approach","author":"Liu","year":"2019"},{"issue":"1","key":"10.1016\/j.ipm.2026.105041_b20","doi-asserted-by":"crossref","DOI":"10.1016\/j.ipm.2023.103546","article-title":"Multi-granularity cross-modal representation learning for named entity recognition on social media","volume":"61","author":"Liu","year":"2024","journal-title":"Information Processing & Management"},{"key":"10.1016\/j.ipm.2026.105041_b21","series-title":"Zero-shot entity linking by reading entity descriptions","author":"Logeswaran","year":"2019"},{"key":"10.1016\/j.ipm.2026.105041_b22","series-title":"Findings of the association for computational linguistics: ACL 2024","first-page":"7559","article-title":"Trust in internal or external knowledge? Generative multi-modal entity linking with knowledge retriever","author":"Long","year":"2024"},{"key":"10.1016\/j.ipm.2026.105041_b23","series-title":"International conference on learning representations","article-title":"Decoupled weight decay regularization","author":"Loshchilov","year":"2019"},{"key":"10.1016\/j.ipm.2026.105041_b24","series-title":"Proceedings of the 29th ACM SIGKDD conference on knowledge discovery and data mining","first-page":"1583","article-title":"Multi-grained multimodal interaction network for entity linking","author":"Luo","year":"2023"},{"key":"10.1016\/j.ipm.2026.105041_b25","series-title":"Findings of the association for computational linguistics: ACL 2025","first-page":"17122","article-title":"VP-MEL: Visual prompts guided multimodal entity linking","author":"Mi","year":"2025"},{"key":"10.1016\/j.ipm.2026.105041_b26","doi-asserted-by":"crossref","unstructured":"Moon, S., Neves, L., & Carvalho, V. (2018). Multimodal named entity disambiguation for noisy social media posts. In Proceedings of the 56th annual meeting of the association for computational linguistics (volume 1: long papers) (pp. 2000\u20132008).","DOI":"10.18653\/v1\/P18-1186"},{"key":"10.1016\/j.ipm.2026.105041_b27","series-title":"ChatGPT","author":"OpenAI","year":"2025"},{"key":"10.1016\/j.ipm.2026.105041_b28","series-title":"Advances in neural information processing systems","article-title":"PyTorch: An imperative style, high-performance deep learning library","volume":"Vol. 32","author":"Paszke","year":"2019"},{"issue":"3","key":"10.1016\/j.ipm.2026.105041_b29","doi-asserted-by":"crossref","first-page":"1625","DOI":"10.1109\/JBHI.2024.3386815","article-title":"Multimodal drug target binding affinity prediction using graph local substructure","volume":"29","author":"Peng","year":"2024","journal-title":"IEEE Journal of Biomedical and Health Informatics"},{"key":"10.1016\/j.ipm.2026.105041_b30","series-title":"PGMEL: Policy gradient-based generative adversarial network for multimodal entity linking","author":"Pooja","year":"2025"},{"key":"10.1016\/j.ipm.2026.105041_b31","series-title":"International conference on machine learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.ipm.2026.105041_b32","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2025.103081","article-title":"Dylas: A dynamic label alignment strategy for large-scale multi-label text classification","volume":"120","author":"Ren","year":"2025","journal-title":"Information Fusion"},{"issue":"2","key":"10.1016\/j.ipm.2026.105041_b33","doi-asserted-by":"crossref","first-page":"443","DOI":"10.1109\/TKDE.2014.2327028","article-title":"Entity linking with a knowledge base: Issues, techniques, and solutions","volume":"27","author":"Shen","year":"2014","journal-title":"IEEE Transactions on Knowledge and Data Engineering"},{"key":"10.1016\/j.ipm.2026.105041_b34","doi-asserted-by":"crossref","unstructured":"Shi, S., Xu, Z., Hu, B. Zhang, M. (2024). Generative multimodal entity linking. In Proceedings of the 2024 joint international conference on computational linguistics, language resources and evaluation (LREC-cOLING 2024) (pp. 7654\u20137665).","DOI":"10.63317\/48465grj4cjm"},{"key":"10.1016\/j.ipm.2026.105041_b35","series-title":"Chinese conference on pattern recognition and computer vision","first-page":"187","article-title":"Dim: Dynamic integration of multimodal entity linking with large language model","author":"Song","year":"2024"},{"key":"10.1016\/j.ipm.2026.105041_b36","series-title":"DWE+: Dual-way matching enhanced framework for multimodal entity linking","author":"Song","year":"2024"},{"key":"10.1016\/j.ipm.2026.105041_b37","doi-asserted-by":"crossref","unstructured":"Song, S., Zhao, S., Wang, C., Yan, T., Li, S., Mao, X., & Wang, M. (2024). A dual-way enhanced framework from text matching point of view for multimodal entity linking. vol. 38, In Proceedings of the AAAI conference on artificial intelligence (pp. 19008\u201319016).","DOI":"10.1609\/aaai.v38i17.29867"},{"key":"10.1016\/j.ipm.2026.105041_b38","series-title":"Findings of the association for computational linguistics ACL 2024","first-page":"816","article-title":"MELOV: Multimodal entity linking with optimized visual features in latent space","author":"Sui","year":"2024"},{"issue":"10","key":"10.1016\/j.ipm.2026.105041_b39","doi-asserted-by":"crossref","first-page":"78","DOI":"10.1145\/2629489","article-title":"Wikidata: a free collaborative knowledgebase","volume":"57","author":"Vrande\u010di\u0107","year":"2014","journal-title":"Communications of the ACM"},{"key":"10.1016\/j.ipm.2026.105041_b40","series-title":"Proceedings of the 60th annual meeting of the association for computational linguistics (volume 1: long papers)","first-page":"4785","article-title":"WikiDiverse: A multimodal entity linking dataset with diversified contextual topics and entity types","author":"Wang","year":"2022"},{"key":"10.1016\/j.ipm.2026.105041_b41","doi-asserted-by":"crossref","DOI":"10.1016\/j.bdr.2020.100159","article-title":"Richpedia: a large-scale, comprehensive multi-modal knowledge graph","volume":"22","author":"Wang","year":"2020","journal-title":"Big Data Research"},{"key":"10.1016\/j.ipm.2026.105041_b42","doi-asserted-by":"crossref","unstructured":"Wang, P., Wu, J., & Chen, X. (2022). Multimodal entity linking with gated hierarchical fusion and contrastive training. In Proceedings of the 45th international ACM SIGIR conference on research and development in information retrieval (pp. 938\u2013948).","DOI":"10.1145\/3477495.3531867"},{"issue":"3","key":"10.1016\/j.ipm.2026.105041_b43","doi-asserted-by":"crossref","DOI":"10.1016\/j.ipm.2025.104507","article-title":"Deepmel: A multi-agent collaboration framework for multimodal entity linking","volume":"63","author":"Wang","year":"2026","journal-title":"Information Processing & Management"},{"key":"10.1016\/j.ipm.2026.105041_b44","series-title":"Proceedings of the 2020 conference on empirical methods in natural language processing","first-page":"6397","article-title":"Scalable zero-shot entity linking with dense entity retrieval","author":"Wu","year":"2020"},{"key":"10.1016\/j.ipm.2026.105041_b45","doi-asserted-by":"crossref","unstructured":"Xing, S., Zhao, F., Wu, Z., Li, C., Zhang, J., & Dai, X. (2023). Drin: Dynamic relation interactive network for multimodal entity linking. In Proceedings of the 31st ACM international conference on multimedia (pp. 3599\u20133608).","DOI":"10.1145\/3581783.3612575"},{"key":"10.1016\/j.ipm.2026.105041_b46","series-title":"CCF international conference on natural language processing and Chinese computing","first-page":"146","article-title":"An adaptive semantic-aware fusion method for multimodal entity linking","author":"Xu","year":"2025"},{"issue":"3","key":"10.1016\/j.ipm.2026.105041_b47","doi-asserted-by":"crossref","DOI":"10.1016\/j.ipm.2024.104048","article-title":"Graph structure prefix injection transformer for multi-modal entity alignment","volume":"62","author":"Zhang","year":"2025","journal-title":"Information Processing & Management"},{"key":"10.1016\/j.ipm.2026.105041_b48","series-title":"Findings of the association for computational linguistics ACL 2024","first-page":"4103","article-title":"Optimal transport guided correlation assignment for multimodal entity linking","author":"Zhang","year":"2024"},{"issue":"1","key":"10.1016\/j.ipm.2026.105041_b49","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1162\/dint_a_00114","article-title":"Visual entity linking via multi-modal learning","volume":"4","author":"Zheng","year":"2022","journal-title":"Data Intelligence"},{"key":"10.1016\/j.ipm.2026.105041_b50","series-title":"Advanced intelligent computing technology and applications","first-page":"478","article-title":"Figmel: A multimodal entity linking framework with large language models and fine-grained semantic class","author":"Zhu","year":"2025"}],"container-title":["Information Processing &amp; Management"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0306457326004322?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0306457326004322?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T13:34:12Z","timestamp":1783690452000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0306457326004322"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2027,1]]},"references-count":50,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2027,1]]}},"alternative-id":["S0306457326004322"],"URL":"https:\/\/doi.org\/10.1016\/j.ipm.2026.105041","relation":{},"ISSN":["0306-4573"],"issn-type":[{"value":"0306-4573","type":"print"}],"subject":[],"published":{"date-parts":[[2027,1]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Dual-Tower Multimodal Entity Linking with Deep Interaction","name":"articletitle","label":"Article Title"},{"value":"Information Processing & Management","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.ipm.2026.105041","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"105041"}}