{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T20:38:24Z","timestamp":1783543104109,"version":"3.55.0"},"reference-count":59,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,7,11]],"date-time":"2025-07-11T00:00:00Z","timestamp":1752192000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2025,7,11]],"date-time":"2025-07-11T00:00:00Z","timestamp":1752192000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Intell Syst"],"DOI":"10.1007\/s44196-025-00820-9","type":"journal-article","created":{"date-parts":[[2025,7,11]],"date-time":"2025-07-11T12:03:09Z","timestamp":1752235389000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":10,"title":["MM-HGNN: Multimodal Representation Learning Heterogeneous Graph Neural Network"],"prefix":"10.1007","volume":"18","author":[{"given":"Khalil","family":"Bachiri","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ali","family":"Yahyaouy","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Maria","family":"Malek","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Nicoleta","family":"Rogovschi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,7,11]]},"reference":[{"key":"820_CR1","doi-asserted-by":"publisher","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 770\u2013778 (2016). https:\/\/doi.org\/10.1109\/CVPR.2016.90. ISSN: 1063-6919. Accessed 1 Sept 2024","DOI":"10.1109\/CVPR.2016.90"},{"key":"820_CR2","doi-asserted-by":"publisher","unstructured":"Reimers, N., Gurevych, I.: Sentence-BERT: sentence embeddings using Siamese BERT-networks (2019). https:\/\/doi.org\/10.48550\/arXiv.1908.10084. arXiv:1908.10084 [cs]. Accessed 1 Sept 2024","DOI":"10.48550\/arXiv.1908.10084"},{"key":"820_CR3","doi-asserted-by":"publisher","unstructured":"Hershey, S., Chaudhuri, S., Ellis, D.P.W., Gemmeke, J.F., Jansen, A., Moore, R.C., Plakal, M., Platt, D., Saurous, R.A., Seybold, B., Slaney, M., Weiss, R.J., Wilson, K.: CNN architectures for large-scale audio classification (2017). https:\/\/doi.org\/10.48550\/arXiv.1609.09430. arXiv:1609.09430 [cs, stat]. Accessed 1 Sept 2024","DOI":"10.48550\/arXiv.1609.09430"},{"issue":"8","key":"820_CR4","doi-asserted-by":"publisher","first-page":"2939","DOI":"10.1007\/s00371-021-02166-7","volume":"38","author":"K Bayoudh","year":"2022","unstructured":"Bayoudh, K., Knani, R., Hamdaoui, F., Mtibaa, A.: A survey on deep multimodal learning for computer vision: advances, trends, applications, and datasets. Vis. Comput. 38(8), 2939\u20132970 (2022). https:\/\/doi.org\/10.1007\/s00371-021-02166-7","journal-title":"Vis. Comput."},{"key":"820_CR5","doi-asserted-by":"publisher","unstructured":"He, R., McAuley, J.: VBPR: visual Bayesian personalized ranking from implicit feedback (2015). https:\/\/doi.org\/10.48550\/arXiv.1510.01784. arXiv:1510.01784 [cs]. Accessed 1 Sept 2024","DOI":"10.48550\/arXiv.1510.01784"},{"key":"820_CR6","doi-asserted-by":"publisher","unstructured":"Bachiri, K., Boufares, F., Malek, M., Rogovschi, N., Yahyaouy, A.: Multi-view clustering using sparse non-negative matrix factorization for recommendation systems. In: 2023 International Conference on Machine Learning and Applications (ICMLA), pp. 1287\u20131294 (2023). https:\/\/doi.org\/10.1109\/ICMLA58977.2023.00194. ISSN: 1946-0759. Accessed 22 Oct 2024","DOI":"10.1109\/ICMLA58977.2023.00194"},{"issue":"9","key":"820_CR7","doi-asserted-by":"publisher","first-page":"4128","DOI":"10.1109\/TIP.2017.2710635","volume":"26","author":"R Hong","year":"2017","unstructured":"Hong, R., Li, L., Cai, J., Tao, D., Wang, M., Tian, Q.: Coherent semantic-visual indexing for large-scale image retrieval in the cloud. IEEE Trans. Image Process. 26(9), 4128\u20134138 (2017). https:\/\/doi.org\/10.1109\/TIP.2017.2710635","journal-title":"IEEE Trans. Image Process."},{"issue":"3","key":"820_CR8","doi-asserted-by":"publisher","first-page":"1235","DOI":"10.1109\/TIP.2018.2875363","volume":"28","author":"M Liu","year":"2019","unstructured":"Liu, M., Nie, L., Wang, X., Tian, Q., Chen, B.: Online data organizer: micro-video categorization by structure-guided multimodal dictionary learning. IEEE Trans. Image Process. 28(3), 1235\u20131247 (2019). https:\/\/doi.org\/10.1109\/TIP.2018.2875363","journal-title":"IEEE Trans. Image Process."},{"key":"820_CR9","doi-asserted-by":"publisher","unstructured":"Sun, R., Cao, X., Zhao, Y., Wan, J., Zhou, K., Zhang, F., Wang, Z., Zheng, K.: Multi-modal knowledge graphs for recommender systems. In: Proceedings of the 29th ACM International Conference on Information & Knowledge Management. CIKM \u201920, pp. 1405\u20131414. Association for Computing Machinery, New York (2020). https:\/\/doi.org\/10.1145\/3340531.3411947. Accessed 31 Aug 2024","DOI":"10.1145\/3340531.3411947"},{"key":"820_CR10","doi-asserted-by":"publisher","first-page":"9008","DOI":"10.1109\/TMM.2024.3384678","volume":"26","author":"Z Song","year":"2024","unstructured":"Song, Z., Hu, Z., Zhou, Y., Zhao, Y., Hong, R., Wang, M.: Embedded heterogeneous attention transformer for cross-lingual image captioning. IEEE Trans. Multimed. 26, 9008\u20139020 (2024). https:\/\/doi.org\/10.1109\/TMM.2024.3384678","journal-title":"IEEE Trans. Multimed."},{"issue":"5","key":"820_CR11","doi-asserted-by":"publisher","first-page":"262","DOI":"10.1007\/s00530-024-01470-1","volume":"30","author":"Y Zhang","year":"2024","unstructured":"Zhang, Y., Song, Z., Hu, Z.: Exploring coherence from heterogeneous representations for OCR image captioning. Multimed. Syst. 30(5), 262 (2024). https:\/\/doi.org\/10.1007\/s00530-024-01470-1","journal-title":"Multimed. Syst."},{"issue":"2","key":"820_CR12","doi-asserted-by":"publisher","first-page":"423","DOI":"10.1109\/TPAMI.2018.2798607","volume":"41","author":"T Baltru\u0161aitis","year":"2019","unstructured":"Baltru\u0161aitis, T., Ahuja, C., Morency, L.-P.: Multimodal machine learning: a survey and taxonomy. IEEE Trans. Pattern Anal. Mach. Intell. 41(2), 423\u2013443 (2019). https:\/\/doi.org\/10.1109\/TPAMI.2018.2798607","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"12","key":"820_CR13","doi-asserted-by":"publisher","first-page":"5659","DOI":"10.1109\/TIP.2015.2487860","volume":"24","author":"C Hong","year":"2015","unstructured":"Hong, C., Yu, J., Wan, J., Tao, D., Wang, M.: Multimodal deep autoencoder for human pose recovery. IEEE Trans. Image Process. 24(12), 5659\u20135670 (2015). https:\/\/doi.org\/10.1109\/TIP.2015.2487860","journal-title":"IEEE Trans. Image Process."},{"key":"820_CR14","doi-asserted-by":"publisher","unstructured":"Khattar, D., Goud, J.S., Gupta, M., Varma, V.: MVAE: multimodal variational autoencoder for fake news detection. In: The World Wide Web Conference. WWW \u201919, pp. 2915\u20132921. Association for Computing Machinery, New York (2019). https:\/\/doi.org\/10.1145\/3308558.3313552. Accessed 30 Aug 2024","DOI":"10.1145\/3308558.3313552"},{"key":"820_CR15","doi-asserted-by":"publisher","unstructured":"Mao, J., Xu, J., Jing, Y., Yuille, A.: Training and evaluating multimodal word embeddings with large-scale web annotated images (2016). https:\/\/doi.org\/10.48550\/arXiv.1611.08321. arXiv:1611.08321 [cs]. Accessed 31 Aug 2024","DOI":"10.48550\/arXiv.1611.08321"},{"key":"820_CR16","doi-asserted-by":"publisher","unstructured":"Baltru\u0161aitis, T., Ahuja, C., Morency, L.-P.: Multimodal machine learning: a survey and taxonomy (2017). https:\/\/doi.org\/10.48550\/arXiv.1705.09406. arXiv:1705.09406 [cs]. Accessed 31 Aug 2024","DOI":"10.48550\/arXiv.1705.09406"},{"key":"820_CR17","doi-asserted-by":"publisher","unstructured":"Huang, Y., Lin, J., Zhou, C., Yang, H., Huang, L.: Modality competition: what makes joint training of multi-modal network fail in deep learning? (Provably) (2022). https:\/\/doi.org\/10.48550\/arXiv.2203.12221. arXiv:2203.12221 [cs]. Accessed 31 Aug 2024","DOI":"10.48550\/arXiv.2203.12221"},{"key":"820_CR18","doi-asserted-by":"publisher","unstructured":"Bachiri, K., Malek, M., Yahyaouy, A., Rogovschi, N.: Adaptive subgraph feature extraction for explainable multi-modal learning. In: 2024 International Conference on Intelligent Systems and Computer Vision (ISCV), pp. 1\u20137 (2024). https:\/\/doi.org\/10.1109\/ISCV60512.2024.10620106. ISSN: 2768-0754. Accessed 22 Oct 2024","DOI":"10.1109\/ISCV60512.2024.10620106"},{"issue":"10","key":"820_CR19","doi-asserted-by":"publisher","first-page":"12113","DOI":"10.1109\/TPAMI.2023.3275156","volume":"45","author":"P Xu","year":"2023","unstructured":"Xu, P., Zhu, X., Clifton, D.A.: Multimodal learning with transformers: a survey. IEEE Trans. Pattern Anal. Mach. Intell. 45(10), 12113\u201312132 (2023). https:\/\/doi.org\/10.1109\/TPAMI.2023.3275156","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"820_CR20","doi-asserted-by":"publisher","unstructured":"Javaloy, A., Meghdadi, M., Valera, I.: Mitigating modality collapse in multimodal VAEs via impartial optimization(2022). https:\/\/doi.org\/10.48550\/arXiv.2206.04496. arXiv:2206.04496 [cs]. Accessed 31 Aug 2024","DOI":"10.48550\/arXiv.2206.04496"},{"key":"820_CR21","doi-asserted-by":"publisher","unstructured":"Ma, M., Ren, J., Zhao, L., Tulyakov, S., Wu, C., Peng, X.: SMIL: multimodal learning with severely missing modality (2021). https:\/\/doi.org\/10.48550\/arXiv.2103.05677. arXiv:2103.05677 [cs]. Accessed 31 Aug 2024","DOI":"10.48550\/arXiv.2103.05677"},{"key":"820_CR22","doi-asserted-by":"publisher","unstructured":"Poklukar, P., Vasco, M., Yin, H., Melo, F.S., Paiva, A., Kragic, D.: Geometric Multimodal contrastive representation learning (2022). https:\/\/doi.org\/10.48550\/arXiv.2202.03390. arXiv:2202.03390 [cs]. Accessed 31 Aug 2024","DOI":"10.48550\/arXiv.2202.03390"},{"key":"820_CR23","doi-asserted-by":"publisher","unstructured":"Wei, Y., Wang, X., Nie, L., He, X., Chua, T.-S.: Graph-refined convolutional network for multimedia recommendation with implicit feedback. In: Proceedings of the 28th ACM International Conference on Multimedia. MM \u201920, pp. 3541\u20133549. Association for Computing Machinery, New York (2020). https:\/\/doi.org\/10.1145\/3394171.3413556. Accessed 31 Aug 2024","DOI":"10.1145\/3394171.3413556"},{"key":"820_CR24","doi-asserted-by":"publisher","unstructured":"Wei, Y., Wang, X., Nie, L., He, X., Hong, R., Chua, T.-S.: MMGCN: Multi-modal graph convolution network for personalized recommendation of micro-video. In: Proceedings of the 27th ACM International Conference on Multimedia. MM \u201919, pp. 1437\u20131445. Association for Computing Machinery, New York (2019). https:\/\/doi.org\/10.1145\/3343031.3351034. Accessed 31 Aug 2024","DOI":"10.1145\/3343031.3351034"},{"key":"820_CR25","doi-asserted-by":"publisher","unstructured":"Veli\u010dkovi\u0107, P., Cucurull, G., Casanova, A., Romero, A., Li\u00f2, P., Bengio, Y.: Graph attention networks (2018). https:\/\/doi.org\/10.48550\/arXiv.1710.10903. arXiv:1710.10903 [cs, stat]. Accessed 24 Aug 2024","DOI":"10.48550\/arXiv.1710.10903"},{"key":"820_CR26","doi-asserted-by":"publisher","unstructured":"Hu, J., Liu, Y., Zhao, J., Jin, Q.: MMGCN: multimodal fusion via deep graph convolution network for emotion recognition in conversation (2021). https:\/\/doi.org\/10.48550\/arXiv.2107.06779. arXiv:2107.06779 [cs, eess]. Accessed 1 Sept 2024","DOI":"10.48550\/arXiv.2107.06779"},{"key":"820_CR27","doi-asserted-by":"publisher","unstructured":"Wang, Y., Qian, S., Hu, J., Fang, Q., Xu, C.: Fake news detection via knowledge-driven multimodal graph convolutional networks. In: Proceedings of the 2020 International Conference on Multimedia Retrieval. ICMR \u201920, pp. 540\u2013547. Association for Computing Machinery, New York (2020). https:\/\/doi.org\/10.1145\/3372278.3390713. Accessed 1 Sept 2024","DOI":"10.1145\/3372278.3390713"},{"issue":"2","key":"820_CR28","doi-asserted-by":"publisher","first-page":"20","DOI":"10.1145\/2481244.2481248","volume":"14","author":"Y Sun","year":"2013","unstructured":"Sun, Y., Han, J.: Mining heterogeneous information networks: a structural analysis approach. SIGKDD Explor. Newsl. 14(2), 20\u201328 (2013). https:\/\/doi.org\/10.1145\/2481244.2481248","journal-title":"SIGKDD Explor. Newsl."},{"issue":"11","key":"820_CR29","doi-asserted-by":"publisher","first-page":"992","DOI":"10.14778\/3402707.3402736","volume":"4","author":"Y Sun","year":"2011","unstructured":"Sun, Y., Han, J., Yan, X., Yu, P.S., Wu, T.: PathSim: meta path-based top-K similarity search in heterogeneous information networks. Proc. VLDB Endow. 4(11), 992\u20131003 (2011). https:\/\/doi.org\/10.14778\/3402707.3402736","journal-title":"Proc. VLDB Endow."},{"key":"820_CR30","doi-asserted-by":"publisher","unstructured":"Huang, Z., Zheng, Y., Cheng, R., Sun, Y., Mamoulis, N., Li, X.: Meta structure: computing relevance in large heterogeneous information networks. In: Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery And Data Mining. KDD \u201916, pp. 1595\u20131604. Association for Computing Machinery, New York (2016). https:\/\/doi.org\/10.1145\/2939672.2939815. Accessed 10 Sept 2024","DOI":"10.1145\/2939672.2939815"},{"key":"820_CR31","doi-asserted-by":"publisher","unstructured":"Cui, P., Wang, X., Pei, J., Zhu, W.: A survey on network embedding (2017). https:\/\/doi.org\/10.48550\/arXiv.1711.08752. arXiv:1711.08752 [cs]. Accessed 10 Sept 2024","DOI":"10.48550\/arXiv.1711.08752"},{"key":"820_CR32","doi-asserted-by":"publisher","unstructured":"Dong, Y., Chawla, N.V., Swami, A.: metapath2vec: scalable representation learning for heterogeneous networks. In: Proceedings of the 23rd ACM SIGKDD International Conference on Knowledge Discovery And Data Mining. KDD \u201917, pp. 135\u2013144. Association for Computing Machinery, New York (2017). https:\/\/doi.org\/10.1145\/3097983.3098036. Accessed 10 Sept 2024","DOI":"10.1145\/3097983.3098036"},{"key":"820_CR33","doi-asserted-by":"publisher","unstructured":"Shi, C., Hu, B., Zhao, W.X., Yu, P.S.: Heterogeneous information network embedding for recommendation (2017). https:\/\/doi.org\/10.48550\/arXiv.1711.10730. arXiv:1711.10730 [cs]. Accessed 10 Sept 2024","DOI":"10.48550\/arXiv.1711.10730"},{"key":"820_CR34","doi-asserted-by":"publisher","unstructured":"Perozzi, B., Al-Rfou, R., Skiena, S.: DeepWalk: online learning of social representations (2014). https:\/\/doi.org\/10.1145\/2623330.2623732. arXiv:1403.6652v2. Accessed 10 Sept 2024","DOI":"10.1145\/2623330.2623732"},{"key":"820_CR35","doi-asserted-by":"publisher","unstructured":"Wang, X., Ji, H., Shi, C., Wang, B., Ye, Y., Cui, P., Yu, P.S.: Heterogeneous graph attention network. In: The World Wide Web Conference. WWW \u201919, pp. 2022\u20132032. Association for Computing Machinery, New York (2019). https:\/\/doi.org\/10.1145\/3308558.3313562. Accessed 10 Sept 2024","DOI":"10.1145\/3308558.3313562"},{"key":"820_CR36","unstructured":"Ngiam, J., Khosla, A., Kim, M., Nam, J., Lee, H., Ng, A.: Multimodal deep learning. In: Proceedings of the 28th International Conference on Machine Learning, ICML 2011, p. 696 (2011)"},{"key":"820_CR37","doi-asserted-by":"publisher","unstructured":"Bengio, Y., Courville, A., Vincent, P.: Representation learning: a review and new perspectives (2014). https:\/\/doi.org\/10.48550\/arXiv.1206.5538. arXiv:1206.5538 [cs]. Accessed 3 Sept 2024","DOI":"10.48550\/arXiv.1206.5538"},{"key":"820_CR38","unstructured":"Srivastava, N., Salakhutdinov, R.R.: Multimodal learning with deep Boltzmann machines. In: Advances in Neural Information Processing Systems, vol. 25. Curran Associates, Inc. (2012)"},{"issue":"7","key":"820_CR39","doi-asserted-by":"publisher","first-page":"1527","DOI":"10.1162\/neco.2006.18.7.1527","volume":"18","author":"GE Hinton","year":"2006","unstructured":"Hinton, G.E., Osindero, S., Teh, Y.-W.: A fast learning algorithm for deep belief nets. Neural Comput. 18(7), 1527\u20131554 (2006). https:\/\/doi.org\/10.1162\/neco.2006.18.7.1527","journal-title":"Neural Comput."},{"key":"820_CR40","doi-asserted-by":"publisher","unstructured":"Kim, Y., Lee, H., Provost, E.M.: Deep learning for robust feature generation in audiovisual emotion recognition. In: 2013 IEEE International Conference on Acoustics, Speech and Signal Processing, pp. 3687\u20133691 (2013). https:\/\/doi.org\/10.1109\/ICASSP.2013.6638346. ISSN: 2379-190X. Accessed 4 Sept 2024","DOI":"10.1109\/ICASSP.2013.6638346"},{"key":"820_CR41","doi-asserted-by":"publisher","unstructured":"Wu, D., Shao, L.: Multimodal dynamic networks for gesture recognition. In: Proceedings of the 22nd ACM International Conference on Multimedia. MM \u201914, pp. 945\u2013948. Association for Computing Machinery, New York (2014). https:\/\/doi.org\/10.1145\/2647868.2654969. Accessed 3 Sept 2024","DOI":"10.1145\/2647868.2654969"},{"issue":"4","key":"820_CR42","doi-asserted-by":"publisher","first-page":"1229","DOI":"10.1109\/TCYB.2017.2685625","volume":"48","author":"A Mandal","year":"2018","unstructured":"Mandal, A., Maji, P.: FaRoC: fast and robust supervised canonical correlation analysis for multimodal omics data. IEEE Trans. Cybern. 48(4), 1229\u20131241 (2018). https:\/\/doi.org\/10.1109\/TCYB.2017.2685625","journal-title":"IEEE Trans. Cybern."},{"key":"820_CR43","doi-asserted-by":"publisher","unstructured":"Yu, Y., Tang, S., Aizawa, K., Aizawa, A.: Category-based deep CCA for fine-grained venue discovery from multimodal data (2018). https:\/\/doi.org\/10.48550\/arXiv.1805.02997. arXiv:1805.02997 [cs]. Accessed 4 Sept 2024","DOI":"10.48550\/arXiv.1805.02997"},{"issue":"5","key":"820_CR44","doi-asserted-by":"publisher","first-page":"1317","DOI":"10.1109\/TMM.2018.2875510","volume":"21","author":"N Elmadany","year":"2019","unstructured":"Elmadany, N., He, Y., Guan, L.: Multimodal learning for human action recognition via bimodal\/multimodal hybrid centroid canonical correlation analysis. IEEE Trans. Multimed. 21(5), 1317\u20131331 (2019). https:\/\/doi.org\/10.1109\/TMM.2018.2875510","journal-title":"IEEE Trans. Multimed."},{"key":"820_CR45","doi-asserted-by":"publisher","unstructured":"Chen, J., Zhang, A.: HGMF: heterogeneous graph-based fusion for multimodal data with incompleteness. In: Proceedings of the 26th ACM SIGKDD International Conference on Knowledge Discovery & Data Mining. KDD \u201920, pp. 1295\u20131305. Association for Computing Machinery, New York (2020). https:\/\/doi.org\/10.1145\/3394486.3403182. Accessed 3 Sept 2024","DOI":"10.1145\/3394486.3403182"},{"key":"820_CR46","doi-asserted-by":"publisher","first-page":"42","DOI":"10.1016\/j.neucom.2020.04.145","volume":"412","author":"J Wang","year":"2020","unstructured":"Wang, J., Hu, J., Qian, S., Fang, Q., Xu, C.: Multimodal graph convolutional networks for high quality content recognition. Neurocomputing 412, 42\u201351 (2020). https:\/\/doi.org\/10.1016\/j.neucom.2020.04.145","journal-title":"Neurocomputing"},{"key":"820_CR47","doi-asserted-by":"publisher","unstructured":"Liu, Y., Sourina, O.: Real-time fractal-based valence level recognition from EEG. In: Gavrilova, M.L., Tan, C.J.K., Kuijper, A. (eds.) Transactions on Computational Science XVIII, pp. 101\u2013120. Springer, Berlin, Heidelberg (2013). https:\/\/doi.org\/10.1007\/978-3-642-38803-3_6","DOI":"10.1007\/978-3-642-38803-3_6"},{"issue":"1","key":"820_CR48","doi-asserted-by":"publisher","first-page":"39","DOI":"10.1109\/TPAMI.2008.52","volume":"31","author":"Z Zeng","year":"2009","unstructured":"Zeng, Z., Pantic, M., Roisman, G.I., Huang, T.S.: A survey of affect recognition methods: audio, visual, and spontaneous expressions. IEEE Trans. Pattern Anal. Mach. Intell. 31(1), 39\u201358 (2009). https:\/\/doi.org\/10.1109\/TPAMI.2008.52","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"3","key":"820_CR49","doi-asserted-by":"publisher","first-page":"43","DOI":"10.1145\/2682899","volume":"47","author":"SK D\u2019mello","year":"2015","unstructured":"D\u2019mello, S.K., Kory, J.: A review and meta-analysis of multimodal affect detection systems. ACM Comput. Surv. 47(3), 43\u201314336 (2015). https:\/\/doi.org\/10.1145\/2682899","journal-title":"ACM Comput. Surv."},{"key":"820_CR50","doi-asserted-by":"crossref","unstructured":"Zadeh, A., Chen, M., Poria, S., Cambria, E., Morency, L.-P.: Tensor Fusion network for multimodal sentiment analysis (2017). https:\/\/arxiv.org\/abs\/1707.07250v1. Accessed 6 Sept 2024","DOI":"10.18653\/v1\/D17-1115"},{"key":"820_CR51","unstructured":"Hou, M., Tang, J., Zhang, J., Kong, W., Zhao, Q.: Deep multimodal multilinear fusion with high-order polynomial pooling. In: Advances in Neural Information Processing Systems, vol. 32. Curran Associates, Inc. (2019)"},{"key":"820_CR52","doi-asserted-by":"publisher","unstructured":"Arevalo, J., Solorio, T., Montes-y-G\u00f3mez, M., Gonz\u00e1lez, F.A.: Gated multimodal units for information fusion (2017). https:\/\/doi.org\/10.48550\/arXiv.1702.01992. arXiv:1702.01992 [cs, stat]. Accessed 8 Sept 2024","DOI":"10.48550\/arXiv.1702.01992"},{"key":"820_CR53","doi-asserted-by":"publisher","unstructured":"Hessel, J., Lee, L.: Does my multimodal model learn cross-modal interactions? It\u2019s harder to tell than you might think! (2020). https:\/\/doi.org\/10.48550\/arXiv.2010.06572. arXiv:2010.06572 [cs]. Accessed 8 Sept 2024","DOI":"10.48550\/arXiv.2010.06572"},{"key":"820_CR54","unstructured":"Kipf, T.N., Welling, M.: Semi-supervised classification with graph convolutional networks (2016). arXiv:1609.02907v4. Accessed 24 Aug 2024"},{"key":"820_CR55","doi-asserted-by":"publisher","unstructured":"Fu, X., Zhang, J., Meng, Z., King, I.: MAGNN: metapath aggregated graph neural network for heterogeneous graph embedding. In: Proceedings of The Web Conference 2020. WWW \u201920, pp. 2331\u20132341. Association for Computing Machinery, New York (2020). https:\/\/doi.org\/10.1145\/3366423.3380297. Accessed 2 Oct 2024","DOI":"10.1145\/3366423.3380297"},{"key":"820_CR56","doi-asserted-by":"publisher","unstructured":"Hong, H., Guo, H., Lin, Y., Yang, X., Li, Z., Ye, J.: An attention-based graph neural network for heterogeneous structural learning (2019). https:\/\/doi.org\/10.48550\/arXiv.1912.10832. arXiv:1912.10832 [cs, stat]. Accessed 2 Oct 2024","DOI":"10.48550\/arXiv.1912.10832"},{"key":"820_CR57","doi-asserted-by":"publisher","unstructured":"Lu, Z., Fang, Y., Yang, C., Shi, C.: Heterogeneous graph transformer with poly-tokenization, p. 2242 (2024). https:\/\/doi.org\/10.24963\/ijcai.2024\/247","DOI":"10.24963\/ijcai.2024\/247"},{"issue":"5","key":"820_CR58","doi-asserted-by":"publisher","first-page":"102277","DOI":"10.1016\/j.ipm.2020.102277","volume":"57","author":"Z Tao","year":"2020","unstructured":"Tao, Z., Wei, Y., Wang, X., He, X., Huang, X., Chua, T.-S.: MGAT: multimodal graph attention network for recommendation. Inf. Process. Manag. 57(5), 102277 (2020). https:\/\/doi.org\/10.1016\/j.ipm.2020.102277","journal-title":"Inf. Process. Manag."},{"key":"820_CR59","doi-asserted-by":"publisher","unstructured":"Zhang, J., Zhu, Y., Liu, Q., Wu, S., Wang, S., Wang, L.: Mining latent structures for multimedia recommendation. In: Proceedings of the 29th ACM International Conference on Multimedia. MM \u201921, pp. 3872\u20133880. Association for Computing Machinery, New York (2021). https:\/\/doi.org\/10.1145\/3474085.3475259. Accessed 20 Dec 2024","DOI":"10.1145\/3474085.3475259"}],"container-title":["International Journal of Computational Intelligence Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s44196-025-00820-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s44196-025-00820-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s44196-025-00820-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,12]],"date-time":"2025-07-12T00:03:04Z","timestamp":1752278584000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s44196-025-00820-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,11]]},"references-count":59,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2025,12]]}},"alternative-id":["820"],"URL":"https:\/\/doi.org\/10.1007\/s44196-025-00820-9","relation":{},"ISSN":["1875-6883"],"issn-type":[{"value":"1875-6883","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,7,11]]},"assertion":[{"value":"6 November 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 March 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"31 March 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 July 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that there is no conflict of interest regarding the publication of this manuscript.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"178"}}