{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T13:16:13Z","timestamp":1778073373638,"version":"3.51.4"},"reference-count":54,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.eswa.2026.132602","type":"journal-article","created":{"date-parts":[[2026,4,28]],"date-time":"2026-04-28T07:03:56Z","timestamp":1777359836000},"page":"132602","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Improving multimodal entity linking through modalities feature disentangling"],"prefix":"10.1016","volume":"325","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-6506-8943","authenticated-orcid":false,"given":"Rong","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ruochun","family":"Jin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-1654-8615","authenticated-orcid":false,"given":"Jintao","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Meng","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-5790-8026","authenticated-orcid":false,"given":"Dong","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiyue","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Cheng","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.132602_bib0001","series-title":"European conference on information retrieval","first-page":"463","article-title":"Multimodal entity linking for tweets","author":"Adjali","year":"2020"},{"key":"10.1016\/j.eswa.2026.132602_bib0002","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2022.116626","article-title":"Multi-modality helps in crisis management: An attention-based deep learning approach of leveraging text for image classification","volume":"195","author":"Ahmad","year":"2022","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132602_bib0003","series-title":"Proceedings of the 28th ACM international conference on information and knowledge management","first-page":"1371","article-title":"Concet: Entity-aware topic classification for open-domain conversational agents","author":"Ahmadvand","year":"2019"},{"key":"10.1016\/j.eswa.2026.132602_bib0004","series-title":"Proceedings of the 22nd ACM international conference on information & knowledge management","first-page":"109","article-title":"Penguins in sweaters, or serendipitous entity search on user-generated content","author":"Bordino","year":"2013"},{"key":"10.1016\/j.eswa.2026.132602_bib0005","unstructured":"Cao, Y., Hou, L., Li, J., & Liu, Z. (2018). Neural collective entity linking. arXiv: 1811.08603[hep-ph]."},{"key":"10.1016\/j.eswa.2026.132602_bib0006","series-title":"Bridge text and knowledge by learning multi-prototype entity mention embedding","author":"Cao","year":"2017"},{"key":"10.1016\/j.eswa.2026.132602_bib0007","series-title":"International conference on machine learning","first-page":"1779","article-title":"Club: A contrastive log-ratio upper bound of mutual information","author":"Cheng","year":"2020"},{"key":"10.1016\/j.eswa.2026.132602_bib0008","doi-asserted-by":"crossref","first-page":"145","DOI":"10.1162\/tacl_a_00129","article-title":"Entity disambiguation with web links","volume":"3","author":"Chisholm","year":"2015","journal-title":"Transactions of the Association for Computational Linguistics"},{"key":"10.1016\/j.eswa.2026.132602_bib0009","doi-asserted-by":"crossref","first-page":"95764","DOI":"10.52202\/079017-3035","article-title":"Contextcite: Attributing model generation to context","volume":"37","author":"Cohen-Wang","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132602_bib0010","unstructured":"van den, O. A., Li, Y., & Vinyals, O. (2018). Representation learning with contrastive predictive coding. arxiv: 1807.03748."},{"key":"10.1016\/j.eswa.2026.132602_bib0011","series-title":"Proceedings of the 2019 conference of the north american chapter of the association for computational linguistics: human language technologies, volume 1 (long and short papers)","first-page":"4171","article-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2019"},{"key":"10.1016\/j.eswa.2026.132602_bib0012","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"18166","article-title":"An empirical study of training end-to-end vision-and-language transformers","author":"Dou","year":"2022"},{"key":"10.1016\/j.eswa.2026.132602_bib0013","doi-asserted-by":"crossref","unstructured":"Eshel, Y., Cohen, N., Radinsky, K., Markovitch, S., Yamada, I., & Levy, O. (2017). Named entity disambiguation for noisy text. arxiv: 1706.09147.","DOI":"10.18653\/v1\/K17-1008"},{"key":"10.1016\/j.eswa.2026.132602_bib0014","series-title":"The world wide web conference","first-page":"438","article-title":"Joint entity linking with deep reinforcement learning","author":"Fang","year":"2019"},{"key":"10.1016\/j.eswa.2026.132602_bib0015","series-title":"Capturing semantic similarity for entity linking with convolutional neural networks","first-page":"1256","author":"Francis-Landau","year":"2016"},{"key":"10.1016\/j.eswa.2026.132602_bib0016","series-title":"Proceedings of the 29th ACM international conference on multimedia","first-page":"993","article-title":"Multimodal entity linking: A new dataset and a baseline","author":"Gan","year":"2021"},{"key":"10.1016\/j.eswa.2026.132602_bib0017","series-title":"Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing","first-page":"6894","article-title":"Simcse: Simple contrastive learning of sentence embeddings","author":"Gao","year":"2021"},{"key":"10.1016\/j.eswa.2026.132602_bib0018","series-title":"Proceedings of the 2017 conference on empirical methods in natural language processing","first-page":"2681","article-title":"Entity linking via joint encoding of types, descriptions, and context","author":"Gupta","year":"2017"},{"key":"10.1016\/j.eswa.2026.132602_bib0019","series-title":"Proceedings of the 2011 conference on empirical methods in natural language processing","first-page":"782","article-title":"Robust disambiguation of named entities in text","author":"Hoffart","year":"2011"},{"key":"10.1016\/j.eswa.2026.132602_bib0020","series-title":"Proceedings of the 31st ACM SIGKDD conference on knowledge discovery and data mining v. 1","first-page":"508","article-title":"Multi-level matching network for multimodal entity linking","author":"Hu","year":"2025"},{"key":"10.1016\/j.eswa.2026.132602_bib0021","series-title":"Proceedings of the 34th ACM international conference on information and knowledge management","first-page":"960","article-title":"Enhancing multimodal entity linking via distillation and multimodal large language models","author":"Huang","year":"2025"},{"key":"10.1016\/j.eswa.2026.132602_bib0022","series-title":"Proceedings of the 26th ACM SIGKDD international conference on knowledge discovery & data mining","first-page":"2553","article-title":"Embedding-based retrieval in facebook search","author":"Huang","year":"2020"},{"key":"10.1016\/j.eswa.2026.132602_bib0023","series-title":"Proceedings of the 2020 conference on empirical methods in natural language processing (EMNLP)","first-page":"6769","article-title":"Dense passage retrieval for open-domain question answering","author":"Karpukhin","year":"2020"},{"key":"10.1016\/j.eswa.2026.132602_bib0024","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"19113","article-title":"Maple: Multi-modal prompt learning","author":"Khattak","year":"2023"},{"key":"10.1016\/j.eswa.2026.132602_bib0025","series-title":"International conference on machine learning","first-page":"5583","article-title":"Vilt: Vision-and-language transformer without convolution or region supervision","author":"Kim","year":"2021"},{"key":"10.1016\/j.eswa.2026.132602_bib0026","series-title":"Proceedings of the 48th international ACM SIGIR conference on research and development in information retrieval","first-page":"989","article-title":"Boosting discriminability for robust multimodal entity linking with visual modality missing","author":"Lao","year":"2025"},{"key":"10.1016\/j.eswa.2026.132602_bib0027","doi-asserted-by":"crossref","unstructured":"Le, P., & Titov, I. (2018). Improving entity linking by modeling latent relations between mentions. arxiv: 1804.10637.","DOI":"10.18653\/v1\/P18-1148"},{"key":"10.1016\/j.eswa.2026.132602_bib0028","first-page":"9694","article-title":"Align before fuse: Vision and language representation learning with momentum distillation","volume":"34","author":"Li","year":"2021","journal-title":"Advances in neural information processing systems"},{"key":"10.1016\/j.eswa.2026.132602_bib0029","series-title":"Proceedings of the 32nd ACM international conference on multimedia","first-page":"7336","article-title":"Generative multimodal data augmentation for low-resource multimodal named entity recognition","author":"Li","year":"2024"},{"key":"10.1016\/j.eswa.2026.132602_bib0030","unstructured":"Liu, Y., Ott, M., Goyal, N., Du, J., Joshi, M., Chen, D., Levy, O., Lewis, M., Zettlemoyer, L., & Stoyanov, V. (2019). Roberta: A robustly optimized bert pretraining approach. arxiv: 1907.11692."},{"key":"10.1016\/j.eswa.2026.132602_bib0031","series-title":"7th international conference on learning representations, ICLR 2019, new orleans, la, usa, may 6\u20139, 2019","article-title":"Decoupled weight decay regularization","author":"Loshchilov","year":"2019"},{"key":"10.1016\/j.eswa.2026.132602_bib0032","series-title":"Proceedings of the 32nd ACM international conference on multimedia","first-page":"9311","article-title":"Bridging gaps in content and knowledge for multimodal entity linking","author":"Luo","year":"2024"},{"key":"10.1016\/j.eswa.2026.132602_bib0033","series-title":"Proceedings of the 29th ACM SIGKDD conference on knowledge discovery and data mining","first-page":"1583","article-title":"Multi-grained multimodal interaction network for entity linking","author":"Luo","year":"2023"},{"key":"10.1016\/j.eswa.2026.132602_bib0034","article-title":"Multimodal contrastive learning with feature disentanglement for sentiment analysis","volume":"238","author":"Ma","year":"2024","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132602_bib0035","doi-asserted-by":"crossref","first-page":"5467","DOI":"10.1109\/TKDE.2025.3580754","article-title":"Multimodal entity linking with dynamic modality selection and interactive prompt learning","volume":"37","author":"Ma","year":"2025","journal-title":"IEEE Transactions on Knowledge and Data Engineering"},{"key":"10.1016\/j.eswa.2026.132602_bib0036","series-title":"Proceedings of the 56th annual meeting of the association for computational linguistics (volume 1: Long papers)","first-page":"2000","article-title":"Multimodal named entity disambiguation for noisy social media posts","author":"Moon","year":"2018"},{"key":"10.1016\/j.eswa.2026.132602_bib0037","series-title":"Icml","first-page":"689","article-title":"Multimodal deep learning","volume":"vol. 11","author":"Ngiam","year":"2011"},{"key":"10.1016\/j.eswa.2026.132602_bib0038","article-title":"Berters: Multimodal representation learning for expert recommendation system with transformer","volume":"143","author":"Nikzad-Khasmakhi","year":"2020","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132602_bib0039","first-page":"8024","article-title":"Pytorch: An imperative style, high-performance deep learning library","volume":"32","author":"Paszke","year":"2019","journal-title":"Advances in neural information processing systems"},{"key":"10.1016\/j.eswa.2026.132602_bib0040","series-title":"International conference on machine learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"issue":"2","key":"10.1016\/j.eswa.2026.132602_bib0041","doi-asserted-by":"crossref","first-page":"443","DOI":"10.1109\/TKDE.2014.2327028","article-title":"Entity linking with a knowledge base: Issues, techniques, and solutions","volume":"27","author":"Shen","year":"2014","journal-title":"IEEE Transactions on Knowledge and Data Engineering"},{"key":"10.1016\/j.eswa.2026.132602_bib0042","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"2818","article-title":"Rethinking the inception architecture for computer vision","author":"Szegedy","year":"2016"},{"key":"10.1016\/j.eswa.2026.132602_bib0043","series-title":"Proceedings of the 2016 conference of the north american chapter of the association for computational linguistics: Human language technologies","first-page":"589","article-title":"Cross-lingual wikification using multilingual embeddings","author":"Tsai","year":"2016"},{"key":"10.1016\/j.eswa.2026.132602_bib0044","series-title":"Proceedings of the conference. association for computational linguistics. meeting","first-page":"6558","article-title":"Multimodal transformer for unaligned multimodal language sequences","volume":"vol. 2019","author":"Tsai","year":"2019"},{"key":"10.1016\/j.eswa.2026.132602_bib0045","series-title":"Proceedings of the 45th international ACM SIGIR conference on research and development in information retrieval","first-page":"938","article-title":"Multimodal entity linking with gated hierarchical fusion and contrastive training","author":"Wang","year":"2022"},{"key":"10.1016\/j.eswa.2026.132602_bib0046","series-title":"Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics","first-page":"4785","article-title":"Wikidiverse: A multimodal entity linking dataset with diversified contextual topics and entity types","author":"Wang","year":"2022"},{"key":"10.1016\/j.eswa.2026.132602_bib0047","series-title":"Proceedings of the web conference 2020","first-page":"1149","article-title":"Dynamic graph convolutional networks for entity linking","author":"Wu","year":"2020"},{"key":"10.1016\/j.eswa.2026.132602_bib0048","doi-asserted-by":"crossref","unstructured":"Wu, L., Petroni, F., Josifoski, M., Riedel, S., & Zettlemoyer, L. (2019). Scalable zero-shot entity linking with dense entity retrieval. arxiv: 1911.03814.","DOI":"10.18653\/v1\/2020.emnlp-main.519"},{"key":"10.1016\/j.eswa.2026.132602_bib0049","series-title":"Proceedings of the 31st ACM international conference on multimedia","first-page":"3599","article-title":"Drin: Dynamic relation interactive network for multimodal entity linking","author":"Xing","year":"2023"},{"key":"10.1016\/j.eswa.2026.132602_bib0050","doi-asserted-by":"crossref","unstructured":"Yamada, I., Shindo, H., Takeda, H., & Takefuji, Y. (2016). Joint learning of the embedding of words and entities for named entity disambiguation. arxiv: 1601.01343.","DOI":"10.18653\/v1\/K16-1025"},{"key":"10.1016\/j.eswa.2026.132602_bib0051","series-title":"International conference on database systems for advanced applications","first-page":"533","article-title":"Attention-based multimodal entity linking with high-quality images","author":"Zhang","year":"2021"},{"key":"10.1016\/j.eswa.2026.132602_bib0052","series-title":"Proceedings of the 47th international ACM SIGIR conference on research and development in information retrieval","first-page":"1883","article-title":"Disentangling id and modality effects for session-based recommendation","author":"Zhang","year":"2024"},{"issue":"1","key":"10.1016\/j.eswa.2026.132602_bib0053","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1162\/dint_a_00114","article-title":"Visual entity linking via multi-modal learning","volume":"4","author":"Zheng","year":"2022","journal-title":"Data Intelligence"},{"issue":"8","key":"10.1016\/j.eswa.2026.132602_bib0054","doi-asserted-by":"crossref","first-page":"2454","DOI":"10.14778\/3742728.3742740","article-title":"OpenMEL: Unsupervised multimodal entity linking using noise-free expanded queries and global coherence","volume":"18","author":"Zhu","year":"2025","journal-title":"Proceedings of the VLDB Endowment"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426015150?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426015150?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T12:38:23Z","timestamp":1778071103000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426015150"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":54,"alternative-id":["S0957417426015150"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132602","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Improving multimodal entity linking through modalities feature disentangling","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132602","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"132602"}}