{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,13]],"date-time":"2026-06-13T00:55:58Z","timestamp":1781312158552,"version":"3.54.1"},"reference-count":59,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100003819","name":"Hubei Province Natural Science Foundation","doi-asserted-by":"publisher","award":["2025AFB653"],"award-info":[{"award-number":["2025AFB653"]}],"id":[{"id":"10.13039\/501100003819","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003819","name":"Hubei Province Natural Science Foundation","doi-asserted-by":"publisher","award":["2025BAB011"],"award-info":[{"award-number":["2025BAB011"]}],"id":[{"id":"10.13039\/501100003819","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62377009"],"award-info":[{"award-number":["62377009"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62207011"],"award-info":[{"award-number":["62207011"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1016\/j.eswa.2026.132296","type":"journal-article","created":{"date-parts":[[2026,4,4]],"date-time":"2026-04-04T15:17:05Z","timestamp":1775315825000},"page":"132296","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["MITRec: Modality intertwined tensorization decomposition for multimodal recommendation"],"prefix":"10.1016","volume":"322","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-6045-5208","authenticated-orcid":false,"given":"Yan","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-3956-0205","authenticated-orcid":false,"given":"Shuai","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-4664-8796","authenticated-orcid":false,"given":"Huibin","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0443-0094","authenticated-orcid":false,"given":"Zhifei","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.132296_bib0001","series-title":"Proceedings of the 40th international ACM SIGIR conference on research and development in information retrieval","first-page":"335","article-title":"Attentive collaborative filtering: multimedia recommendation with item-and component-level attention","author":"Chen","year":"2017"},{"key":"10.1016\/j.eswa.2026.132296_bib0002","series-title":"Proceedings of the 24th ACM international conference on multimedia","first-page":"1018","article-title":"Context-aware image tweet modelling and recommendation","author":"Chen","year":"2016"},{"key":"10.1016\/j.eswa.2026.132296_bib0003","series-title":"Proceedings of the 42nd international ACM SIGIR conference on research and development in information retrieval","first-page":"765","article-title":"Personalized fashion recommendation with visual explanations based on multimodal attention network: towards visually explainable recommendation","author":"Chen","year":"2019"},{"key":"10.1016\/j.eswa.2026.132296_bib0004","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"11828","article-title":"Mod-squad: designing mixtures of experts as modular multi-task learners","author":"Chen","year":"2023"},{"key":"10.1016\/j.eswa.2026.132296_bib0005","article-title":"Cdmr2f: correlation-guided denoising multimodal robust fusion framework for multimodal recommendation","author":"Cheng","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132296_bib0006","article-title":"Discrepancy learning guided hierarchical fusion network for multi-modal recommendation","author":"Dang","year":"2025","journal-title":"Knowledge-Based Systems"},{"key":"10.1016\/j.eswa.2026.132296_bib0007","series-title":"Proceedings of the 2019 conference of the north american chapter of the association for computational linguistics","first-page":"4171","article-title":"Bert: pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2019"},{"key":"10.1016\/j.eswa.2026.132296_bib0008","series-title":"Proceedings of the 28th ACM international conference on multimedia","first-page":"3469","article-title":"How to learn item representation for cold-start multimedia recommendation?","author":"Du","year":"2020"},{"key":"10.1016\/j.eswa.2026.132296_bib0009","series-title":"Proceedings of the 25th ACM international conference on multimedia","first-page":"127","article-title":"A unified personalized video recommendation via dynamic recurrent neural networks","author":"Gao","year":"2017"},{"key":"10.1016\/j.eswa.2026.132296_bib0010","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"8454","article-title":"Lgmrec: local and global graph learning for multimodal recommendation","author":"Guo","year":"2024"},{"key":"10.1016\/j.eswa.2026.132296_bib0011","doi-asserted-by":"crossref","first-page":"67850","DOI":"10.52202\/079017-2167","article-title":"Fusemoe: mixture-of-experts transformers for fleximodal fusion","author":"Han","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132296_bib0012","series-title":"Proceedings of the thirtieth AAAI conference on artificial intelligence","first-page":"144","article-title":"VBPR: Visual bayesian personalized ranking from implicit feedback","author":"He","year":"2016"},{"key":"10.1016\/j.eswa.2026.132296_bib0013","series-title":"Proceedings of the 24th ACM international on conference on information and knowledge management","first-page":"1661","article-title":"Trirank: review-aware explainable recommendation by modeling aspects","author":"He","year":"2015"},{"key":"10.1016\/j.eswa.2026.132296_bib0014","series-title":"Proceedings of the 43rd international ACM SIGIR conference on research and development in information retrieval","first-page":"639","article-title":"Lightgcn: simplifying and powering graph convolution network for recommendation","author":"He","year":"2020"},{"key":"10.1016\/j.eswa.2026.132296_bib0015","series-title":"Proceedings of the 26th international conference on world wide web","first-page":"173","article-title":"Neural collaborative filtering","author":"He","year":"2017"},{"key":"10.1016\/j.eswa.2026.132296_bib0016","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"11790","article-title":"Modality-independent graph neural networks with global transformers for multimodal recommendation","author":"Hu","year":"2025"},{"key":"10.1016\/j.eswa.2026.132296_bib0017","doi-asserted-by":"crossref","first-page":"3281","DOI":"10.1109\/TKDE.2023.3348537","article-title":"Mgdcf: distance learning via markov graph diffusion for neural collaborative filtering","author":"Hu","year":"2024","journal-title":"IEEE Transactions on Knowledge and Data Engineering"},{"key":"10.1016\/j.eswa.2026.132296_bib0018","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"11808","article-title":"Beyond graph convolution: multimodal recommendation with topology-aware mlps","author":"Huang","year":"2025"},{"key":"10.1016\/j.eswa.2026.132296_bib0019","series-title":"Proceedings of the 31st ACM international conference on multimedia","first-page":"6410","article-title":"Pareto invariant representation learning for multimedia recommendation","author":"Huang","year":"2023"},{"key":"10.1016\/j.eswa.2026.132296_bib0020","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2025.128240","article-title":"Graph-based technology recommendation system using GAT-NGCF","author":"Kim","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132296_bib0021","series-title":"Proceedings of the 31st ACM international conference on information & knowledge management","first-page":"993","article-title":"Mario: modality-aware attention and modality-preserving decoders for multimedia recommendation","author":"Kim","year":"2022"},{"key":"10.1016\/j.eswa.2026.132296_bib0022","series-title":"Proceedings of the 17th ACM international conference on web search and data mining, WSDM 2024, merida, mexico, march 4\u20138, 2024","first-page":"332","article-title":"MONET: Modality-embracing graph convolutional network and target-aware attention for multimedia recommendation","author":"Kim","year":"2024"},{"key":"10.1016\/j.eswa.2026.132296_bib0023","series-title":"Proceedings of the 5th international conference on learning representations","article-title":"Semi-supervised classification with graph convolutional networks","author":"Kipf","year":"2017"},{"key":"10.1016\/j.eswa.2026.132296_bib0024","article-title":"Knowledge graph attention network with path rotate encoding for recommendation","author":"Li","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132296_bib0025","series-title":"Proceedings of the 27th ACM international conference on multimedia","first-page":"1526","article-title":"User diverse preference modeling by multimodal attentive metric learning","author":"Liu","year":"2019"},{"key":"10.1016\/j.eswa.2026.132296_bib0026","series-title":"Proceedings of the 40th international ACM SIGIR conference on research and development in information retrieval","first-page":"841","article-title":"Deepstyle: learning user preferences for visual recommendation","author":"Liu","year":"2017"},{"key":"10.1016\/j.eswa.2026.132296_bib0027","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3662738","article-title":"Formalizing multimedia recommendation through multimodal deep learning","author":"Malitesta","year":"2025","journal-title":"ACM Transactions on Recommender Systems"},{"key":"10.1016\/j.eswa.2026.132296_bib0028","series-title":"Proceedings of the 30th ACM international conference on information & knowledge management","first-page":"1253","article-title":"UltraGCN: ultra simplification of graph convolutional networks for recommendation","author":"Mao","year":"2021"},{"key":"10.1016\/j.eswa.2026.132296_bib0029","series-title":"Proceedings of the 38th international ACM SIGIR conference on research and development in information retrieval","first-page":"43","article-title":"Image-based recommendations on styles and substitutes","author":"McAuley","year":"2015"},{"key":"10.1016\/j.eswa.2026.132296_bib0030","first-page":"14200","article-title":"Attention bottlenecks for multimodal fusion","author":"Nagrani","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132296_bib0031","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"12461","article-title":"Seeing beyond noise: joint graph structure evaluation and denoising for multimodal recommendation","author":"Qi","year":"2025"},{"key":"10.1016\/j.eswa.2026.132296_bib0032","series-title":"Proceedings of the twenty-fifth conference on uncertainty in artificial intelligence","first-page":"452","article-title":"BPR: Bayesian personalized ranking from implicit feedback","author":"Rendle","year":"2009"},{"key":"10.1016\/j.eswa.2026.132296_bib0033","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3661821","article-title":"A survey of graph neural networks for social recommender systems","author":"Sharma","year":"2024","journal-title":"ACM Computing Surveys"},{"key":"10.1016\/j.eswa.2026.132296_bib0034","series-title":"Proceedings of the 3rd international conference on learning representations","article-title":"Very deep convolutional networks for large-scale image recognition","author":"Simonyan","year":"2015"},{"key":"10.1016\/j.eswa.2026.132296_bib0035","doi-asserted-by":"crossref","DOI":"10.1155\/2009\/421425","article-title":"A survey of collaborative filtering techniques","author":"Su","year":"2009","journal-title":"Advances in Artificial Intelligence"},{"key":"10.1016\/j.eswa.2026.132296_bib0036","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2024.124714","article-title":"Care: context-aware attention interest redistribution for session-based recommendation","author":"Tong","year":"2024","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132296_bib0037","first-page":"2579","article-title":"Visualizing data using t-SNE","author":"van der","year":"2008","journal-title":"Journal of Machine Learning Research"},{"key":"10.1016\/j.eswa.2026.132296_bib0038","article-title":"Graph convolutional matrix completion","author":"van den Berg","year":"2017","journal-title":"CoRR"},{"key":"10.1016\/j.eswa.2026.132296_bib0039","doi-asserted-by":"crossref","first-page":"1074","DOI":"10.1109\/TMM.2021.3138298","article-title":"DualGNN: dual graph neural network for multimedia recommendation","author":"Wang","year":"2023","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.eswa.2026.132296_bib0040","series-title":"Proceedings of the 26th international conference on world wide web","first-page":"391","article-title":"What your images reveal: exploiting visual contents for point-of-interest recommendation","author":"Wang","year":"2017"},{"key":"10.1016\/j.eswa.2026.132296_bib0041","series-title":"Proceedings of the 42nd international ACM SIGIR conference on research and development in information retrieval","first-page":"165","article-title":"Neural graph collaborative filtering","author":"Wang","year":"2019"},{"key":"10.1016\/j.eswa.2026.132296_bib0042","series-title":"Proceedings of the 27th ACM international conference on multimedia","first-page":"1437","article-title":"Mmgcn: multi-modal graph convolution network for personalized recommendation of micro-video","author":"Wei","year":"2019"},{"key":"10.1016\/j.eswa.2026.132296_bib0043","series-title":"The thirteenth international conference on learning representations","article-title":"Dynamic modeling of patients, modalities and tasks via multi-modal multi-task mixture of experts","author":"Wu","year":"2025"},{"key":"10.1016\/j.eswa.2026.132296_bib0044","series-title":"Proceedings of the 48th international ACM SIGIR conference on research and development in information retrieval","first-page":"1830","article-title":"Cohesion: composite graph convolutional network with dual-stage fusion for multimodal recommendation","author":"Xu","year":"2025"},{"key":"10.1016\/j.eswa.2026.132296_bib0045","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"12908","article-title":"Mentor: multi-level self-supervised learning for multimodal recommendation","author":"Xu","year":"2025"},{"key":"10.1016\/j.eswa.2026.132296_bib0046","doi-asserted-by":"crossref","first-page":"1354","DOI":"10.1109\/TIP.2023.3243521","article-title":"Adaptive feature projection with distribution alignment for deep incomplete multi-view clustering","author":"Xu","year":"2023","journal-title":"IEEE Transactions on Image Processing"},{"key":"10.1016\/j.eswa.2026.132296_bib0047","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","article-title":"Robust multi-view learning via representation fusion of sample-level attention and alignment of simulated perturbation","author":"Xu","year":"2025"},{"key":"10.1016\/j.eswa.2026.132296_bib0048","series-title":"Proceedings of the international conference on learning representations","article-title":"Slmrec: empowering small language models for sequential recommendation","author":"Xu","year":"2025"},{"key":"10.1016\/j.eswa.2026.132296_bib0049","first-page":"741","article-title":"Multi-modal discrete collaborative filtering for efficient cold-start recommendation","author":"Xu","year":"2021","journal-title":"IEEE Transactions on Knowledge and Data Engineering"},{"key":"10.1016\/j.eswa.2026.132296_bib0050","series-title":"Proceedings of the 24th ACM SIGKDD international conference on knowledge discovery & data mining","first-page":"974","article-title":"Graph convolutional neural networks for web-scale recommender systems","author":"Ying","year":"2018"},{"key":"10.1016\/j.eswa.2026.132296_bib0051","series-title":"Proceedings of the 31st ACM international conference on multimedia","first-page":"6576","article-title":"Multi-view graph convolutional network for multimedia recommendation","author":"Yu","year":"2023"},{"key":"10.1016\/j.eswa.2026.132296_bib0052","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2025.114020","article-title":"Cross-modal feature alignment and fusion with contrastive learning in multimodal recommendation","author":"Yuan","year":"2025","journal-title":"Knowledge-Based Systems"},{"key":"10.1016\/j.eswa.2026.132296_bib0053","series-title":"Proceedings of the ACM web conference 2023","first-page":"759","article-title":"ApeGNN: node-wise adaptive aggregation in GNNs for recommendation","author":"Zhang","year":"2023"},{"key":"10.1016\/j.eswa.2026.132296_bib0054","series-title":"Proceedings of the 29th ACM international conference on multimedia","first-page":"3872","article-title":"Mining latent structures for multimedia recommendation","author":"Zhang","year":"2021"},{"key":"10.1016\/j.eswa.2026.132296_bib0055","first-page":"9154","article-title":"Latent structure mining with contrastive modality fusion for multimedia recommendation","author":"Zhang","year":"2022","journal-title":"IEEE Transactions on Knowledge and Data Engineering"},{"key":"10.1016\/j.eswa.2026.132296_bib0056","unstructured":"Zhou, H., Zhou, X., Zeng, Z., Zhang, L., & Shen, Z. (2023a). A comprehensive survey on multimodal recommender systems: taxonomy, evaluation, and future directions. arXiv preprint arXiv: 2302.04473."},{"key":"10.1016\/j.eswa.2026.132296_bib0057","series-title":"Proceedings of the 26th european conference on artificial intelligence","first-page":"3123","article-title":"Enhancing dyadic relations with homogeneous graphs for multimodal recommendation","author":"Zhou","year":"2023"},{"key":"10.1016\/j.eswa.2026.132296_bib0058","series-title":"Proceedings of the 31st ACM international conference on multimedia","first-page":"935","article-title":"A tale of two graphs: freezing and denoising graph structures for multimodal recommendation","author":"Zhou","year":"2023"},{"key":"10.1016\/j.eswa.2026.132296_bib0059","series-title":"Proceedings of the ACM web conference 2023","first-page":"845","article-title":"Bootstrap latent representations for multi-modal recommendation","author":"Zhou","year":"2023"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426012091?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426012091?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,13]],"date-time":"2026-06-13T00:02:19Z","timestamp":1781308939000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426012091"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8]]},"references-count":59,"alternative-id":["S0957417426012091"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132296","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,8]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"MITRec: Modality intertwined tensorization decomposition for multimodal recommendation","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132296","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"132296"}}