{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,7]],"date-time":"2026-03-07T08:04:41Z","timestamp":1772870681566,"version":"3.50.1"},"reference-count":40,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2024,5,9]],"date-time":"2024-05-09T00:00:00Z","timestamp":1715212800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,5,9]],"date-time":"2024-05-09T00:00:00Z","timestamp":1715212800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62277009"],"award-info":[{"award-number":["62277009"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2024,6]]},"DOI":"10.1007\/s00530-024-01310-2","type":"journal-article","created":{"date-parts":[[2024,5,9]],"date-time":"2024-05-09T06:41:57Z","timestamp":1715236917000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["Semantic-wise guidance for efficient multimodal emotion recognition with missing modalities"],"prefix":"10.1007","volume":"30","author":[{"given":"Shuhua","family":"Liu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yixuan","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kehan","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Binshuai","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fengqin","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shihao","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,5,9]]},"reference":[{"key":"1310_CR1","doi-asserted-by":"publisher","unstructured":"Aguilar, G., Rozgic, V., Wang, W., and Wang, C.: Multimodal and multi-view models for emotion recognition. Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics, 991\u20131002. https:\/\/doi.org\/10.18653\/v1\/P19-1095 2019","DOI":"10.18653\/v1\/P19-1095"},{"key":"1310_CR2","doi-asserted-by":"publisher","first-page":"115507","DOI":"10.1016\/j.eswa.2021.115507","volume":"184","author":"KA Ara\u00f1o","year":"2021","unstructured":"Ara\u00f1o, K.A., Orsenigo, C., Soto, M., Vercellis, C.: Multimodal sentiment and emotion recognition in hyperbolic space. Expert Syst. Appl. 184, 115507 (2021). https:\/\/doi.org\/10.1016\/j.eswa.2021.115507","journal-title":"Expert Syst. Appl."},{"key":"1310_CR3","doi-asserted-by":"publisher","first-page":"113723","DOI":"10.1016\/j.eswa.2020.113723","volume":"160","author":"I Baidari","year":"2020","unstructured":"Baidari, I., Honnikoll, N.: Accuracy weighted diversity-based online boosting. Expert Syst. Appl. 160, 113723 (2020). https:\/\/doi.org\/10.1016\/j.eswa.2020.113723","journal-title":"Expert Syst. Appl."},{"issue":"4","key":"1310_CR4","doi-asserted-by":"publisher","first-page":"335","DOI":"10.1007\/s10579-008-9076-6","volume":"42","author":"C Busso","year":"2008","unstructured":"Busso, C., Bulut, M., Lee, C.-C., Kazemzadeh, A., Mower, E., Kim, S., Chang, J.N., Lee, S., Narayanan, S.S.: IEMOCAP: Interactive emotional dyadic motion capture database. Lang. Resour. Eval. 42(4), 335\u2013359 (2008). https:\/\/doi.org\/10.1007\/s10579-008-9076-6","journal-title":"Lang. Resour. Eval."},{"issue":"1","key":"1310_CR5","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1109\/TAFFC.2016.2515617","volume":"8","author":"C Busso","year":"2017","unstructured":"Busso, C., Parthasarathy, S., Burmania, A., AbdelWahab, M., Sadoughi, N., Provost, E.M.: MSP-IMPROV: an acted corpus of dyadic interactions to study emotion perception. IEEE Trans. Affect. Comput. 8(1), 67\u201380 (2017). https:\/\/doi.org\/10.1109\/TAFFC.2016.2515617","journal-title":"IEEE Trans. Affect. Comput."},{"key":"1310_CR6","doi-asserted-by":"publisher","unstructured":"Cai, L., Wang, Z., Gao, H., Shen, D., and Ji, S.: Deep adversarial learning for multi-modality missing data completion. Proceedings of the 24th ACM SIGKDD International Conference on Knowledge Discovery & Data Mining, 1158\u20131166. https:\/\/doi.org\/10.1145\/3219819.32199632018","DOI":"10.1145\/3219819.3219963"},{"key":"1310_CR7","doi-asserted-by":"publisher","unstructured":"Devlin, J., Chang, M.-W., Lee, K., and Toutanova, K.: BERT: pre-training of deep bidirectional transformers for language understanding. Proceedings of the 2019 Conference of the North, 4171\u20134186. https:\/\/doi.org\/10.18653\/v1\/N19-1423 2019","DOI":"10.18653\/v1\/N19-1423"},{"key":"1310_CR8","doi-asserted-by":"publisher","unstructured":"Eyben, F., W\u00f6llmer, M., and Schuller, B.: Opensmile: The munich versatile and fast open-source audio feature extractor. Proceedings of the 18th ACM International Conference on Multimedia, 1459\u20131462. https:\/\/doi.org\/10.1145\/1873951.1874246 2010","DOI":"10.1145\/1873951.1874246"},{"key":"1310_CR9","unstructured":"Gong, P., Liu, J., Zhang, X., Li, X., and Yu, Z.: Circulant-interactive transformer with dimension-aware fusion for multimodal sentiment analysis. 189. 2023"},{"issue":"31\u201332","key":"1310_CR10","doi-asserted-by":"publisher","first-page":"23347","DOI":"10.1007\/s11042-020-09068-1","volume":"79","author":"S Gupta","year":"2020","unstructured":"Gupta, S., Fahad, Md.S., Deepak, A.: Pitch-synchronous single frequency filtering spectrogram for speech emotion recognition. Multimed. Tools Appl. 79(31\u201332), 23347\u201323365 (2020). https:\/\/doi.org\/10.1007\/s11042-020-09068-1","journal-title":"Multimed. Tools Appl."},{"key":"1310_CR11","doi-asserted-by":"publisher","unstructured":"Han, J., Zhang, Z., Ren, Z., & Schuller, B.: Implicit fusion by joint audiovisual training for emotion recognition in mono modality. ICASSP 2019 - 2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), 5861\u20135865. https:\/\/doi.org\/10.1109\/ICASSP.2019.8682773 2019","DOI":"10.1109\/ICASSP.2019.8682773"},{"key":"1310_CR12","doi-asserted-by":"publisher","unstructured":"Hazarika, D., Zimmermann, R., and Poria, S.: MISA: Modality-invariant and -specific representations for multimodal sentiment analysis. Proceedings of the 28th ACM International Conference on Multimedia, 1122\u20131131. https:\/\/doi.org\/10.1145\/3394171.3413678 2020","DOI":"10.1145\/3394171.3413678"},{"key":"1310_CR13","doi-asserted-by":"crossref","unstructured":"Hu, F., Chen, A., Wang, Z., Zhou, F., Dong, J., & Li, X.: Lightweight attentional feature fusion: a new baseline for text-to-video retrieval (arXiv:2112.01832). arXiv. http:\/\/arxiv.org\/abs\/2112.01832 2022","DOI":"10.1007\/978-3-031-19781-9_26"},{"key":"1310_CR14","doi-asserted-by":"publisher","unstructured":"Hu, J., Shen, L., and Sun, G.: Squeeze-and-excitation networks. 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 7132\u20137141 https:\/\/doi.org\/10.1109\/CVPR.2018.00745 2018","DOI":"10.1109\/CVPR.2018.00745"},{"key":"1310_CR15","doi-asserted-by":"publisher","unstructured":"Huang, G., Liu, Z., Van Der Maaten, L., and Weinberger, K. Q.: Densely connected convolutional networks. 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 2261\u20132269. https:\/\/doi.org\/10.1109\/CVPR.2017.243 2017","DOI":"10.1109\/CVPR.2017.243"},{"key":"1310_CR16","doi-asserted-by":"publisher","unstructured":"Kingma, D. P., and Ba, J.: Adam: a method for stochastic optimization. https:\/\/doi.org\/10.48550\/ARXIV.1412.6980 2014","DOI":"10.48550\/ARXIV.1412.6980"},{"key":"1310_CR17","doi-asserted-by":"publisher","unstructured":"Lian, Z., Chen, L., Sun, L., Liu, B., and Tao, J.: GCNet: Graph completion network for incomplete multimodal learning in conversation. IEEE Transactions on Pattern Analysis and Machine Intelligence, 1\u201314. https:\/\/doi.org\/10.1109\/TPAMI.2023.3234553 2023","DOI":"10.1109\/TPAMI.2023.3234553"},{"key":"1310_CR18","doi-asserted-by":"publisher","unstructured":"Liang, J., Li, R., and Jin, Q.: Semi-supervised multi-modal emotion recognition with cross-modal distribution matching. Proceedings of the 28th ACM International Conference on Multimedia, 2852\u20132861. https:\/\/doi.org\/10.1145\/3394171.3413579 2020","DOI":"10.1145\/3394171.3413579"},{"key":"1310_CR19","doi-asserted-by":"publisher","unstructured":"Lin, T.-Y., Dollar, P., Girshick, R., He, K., Hariharan, B., & Belongie, S.: Feature pyramid networks for object detection. 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 936\u2013944. https:\/\/doi.org\/10.1109\/CVPR.2017.106 2017","DOI":"10.1109\/CVPR.2017.106"},{"key":"1310_CR20","doi-asserted-by":"publisher","first-page":"46","DOI":"10.1016\/j.neucom.2022.05.007","volume":"496","author":"J Liu","year":"2022","unstructured":"Liu, J., Wang, H., Sun, M., Wei, Y.: Graph based emotion recognition with attention pooling for variable-length utterances. Neurocomputing 496, 46\u201355 (2022). https:\/\/doi.org\/10.1016\/j.neucom.2022.05.007","journal-title":"Neurocomputing"},{"key":"1310_CR21","doi-asserted-by":"publisher","unstructured":"Liu, Z., Shen, Y., Lakshminarasimhan, V. B., Liang, P. P., Bagher Zadeh, A., & Morency, L.-P.: Efficient low-rank multimodal fusion with modality-specific factors. Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), 2247\u20132256. https:\/\/doi.org\/10.18653\/v1\/P18-1209 2018","DOI":"10.18653\/v1\/P18-1209"},{"key":"1310_CR22","doi-asserted-by":"publisher","unstructured":"Luo, W., Xu, M., & Lai, H.: Multimodal reconstruct and align net for missing modality problem in sentiment analysis. In D.-T. Dang-Nguyen, C. Gurrin, M. Larson, A. F. Smeaton, S. Rudinac, M.-S. Dao, C. Trattner, & P. Chen (Eds.), MultiMedia Modeling (Vol. 13834, pp. 411\u2013422). Springer Nature Switzerland. https:\/\/doi.org\/10.1007\/978-3-031-27818-1_34 2023","DOI":"10.1007\/978-3-031-27818-1_34"},{"key":"1310_CR23","doi-asserted-by":"publisher","unstructured":"Lv, F., Chen, X., Huang, Y., Duan, L., and Lin, G.: Progressive modality reinforcement for human multimodal emotion recognition from unaligned multimodal sequences. 2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), 2554\u20132562. https:\/\/doi.org\/10.1109\/CVPR46437.2021.00258 2021","DOI":"10.1109\/CVPR46437.2021.00258"},{"key":"1310_CR24","doi-asserted-by":"publisher","unstructured":"Mun, J., Cho, M., and Han, B.: Text-guided attention model for image captioning. https:\/\/doi.org\/10.48550\/ARXIV.1612.03557 2016","DOI":"10.48550\/ARXIV.1612.03557"},{"key":"1310_CR25","doi-asserted-by":"publisher","unstructured":"Pan, Z., Luo, Z., Yang, J., and Li, H.: Multi-modal attention for speech emotion recognition. https:\/\/doi.org\/10.48550\/ARXIV.2009.04107 2020","DOI":"10.48550\/ARXIV.2009.04107"},{"key":"1310_CR26","doi-asserted-by":"publisher","unstructured":"Pham, H., Liang, P. P., Manzini, T., Morency, L.-P., and P\u00f3czos, B.: found in translation: learning robust joint representations by cyclic translations between modalities. Proceedings of the AAAI Conference on Artificial Intelligence, 33(01), 6892\u20136899. https:\/\/doi.org\/10.1609\/aaai.v33i01.330168922019 2019","DOI":"10.1609\/aaai.v33i01.330168922019"},{"key":"1310_CR27","doi-asserted-by":"publisher","unstructured":"Poklukar, P., Vasco, M., Yin, H., Melo, F. S., Paiva, A., and Kragic, D.: Geometric multimodal contrastive representation learning. https:\/\/doi.org\/10.48550\/ARXIV.2202.03390 2022","DOI":"10.48550\/ARXIV.2202.03390"},{"key":"1310_CR28","doi-asserted-by":"publisher","first-page":"98","DOI":"10.1016\/j.inffus.2017.02.003","volume":"37","author":"S Poria","year":"2017","unstructured":"Poria, S., Cambria, E., Bajpai, R., Hussain, A.: A review of affective computing: from unimodal analysis to multimodal fusion. Inform Fusion 37, 98\u2013125 (2017). https:\/\/doi.org\/10.1016\/j.inffus.2017.02.003","journal-title":"Inform Fusion"},{"key":"1310_CR29","doi-asserted-by":"publisher","unstructured":"Tang, S., Luo, Z., Nan, G., Baba, J., Yoshikawa, Y., and Ishiguro, H.: Fusion with hierarchical graphs for multimodal emotion recognition. 2022 Asia-Pacific Signal and Information Processing Association Annual Summit and Conference (APSIPA ASC), 1288\u20131296. https:\/\/doi.org\/10.23919\/APSIPAASC55919.2022.9979932 2022","DOI":"10.23919\/APSIPAASC55919.2022.9979932"},{"key":"1310_CR30","doi-asserted-by":"publisher","unstructured":"Tran, L., Liu, X., Zhou, J., and Jin, R.: Missing Modalities Imputation via Cascaded Residual Autoencoder. 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 4971\u20134980. https:\/\/doi.org\/10.1109\/CVPR.2017.528 2017","DOI":"10.1109\/CVPR.2017.528"},{"key":"1310_CR31","doi-asserted-by":"publisher","unstructured":"Tsai, Y.-H. H., Bai, S., Liang, P. P., Kolter, J. Z., Morency, L.-P., and Salakhutdinov, R.: Multimodal transformer for unaligned multimodal language sequences. Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics, 6558\u20136569. https:\/\/doi.org\/10.18653\/v1\/P19-1656 2019","DOI":"10.18653\/v1\/P19-1656"},{"key":"1310_CR32","doi-asserted-by":"publisher","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A. N., Kaiser, L., and Polosukhin, I.: Attention is all you need. https:\/\/doi.org\/10.48550\/ARXIV.1706.03762 2017","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"1310_CR33","doi-asserted-by":"publisher","unstructured":"Wang, Z., Wan, Z., and Wan, X.: TransModality: An end2end fusion method with transformer for multimodal sentiment analysis. Proceedings of The Web Conference 2020, 2514\u20132520. https:\/\/doi.org\/10.1145\/3366423.3380000 2020","DOI":"10.1145\/3366423.3380000"},{"key":"1310_CR34","doi-asserted-by":"publisher","unstructured":"Wu, N., Green, B., Ben, X., and O\u2019Banion, S.: Deep Transformer Models for Time Series Forecasting: The Influenza Prevalence Case. https:\/\/doi.org\/10.48550\/ARXIV.2001.08317 2020","DOI":"10.48550\/ARXIV.2001.08317"},{"key":"1310_CR35","doi-asserted-by":"publisher","unstructured":"Zadeh, A., Chen, M., Poria, S., Cambria, E., and Morency, L.-P.: Tensor fusion network for multimodal sentiment analysis. Proceedings of the 2017 Conference on Empirical Methods in Natural Language Processing, 1103\u20131114. https:\/\/doi.org\/10.18653\/v1\/D17-1115 2017","DOI":"10.18653\/v1\/D17-1115"},{"key":"1310_CR36","doi-asserted-by":"publisher","unstructured":"Zhang, M., Mosbach, M., Adelani, D., Hedderich, M., & Klakow, D.: MCSE: Multimodal contrastive learning of sentence embeddings. Proceedings of the 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, 5959\u20135969. https:\/\/doi.org\/10.18653\/v1\/2022.naacl-main.436 2022","DOI":"10.18653\/v1\/2022.naacl-main.436"},{"key":"1310_CR37","doi-asserted-by":"publisher","unstructured":"Zhao, J., Li, R., and Jin, Q.: Missing modality imagination network for emotion recognition with uncertain missing modalities. Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers), 2608\u20132618. https:\/\/doi.org\/10.18653\/v1\/2021.acl-long.203 2021","DOI":"10.18653\/v1\/2021.acl-long.203"},{"key":"1310_CR38","doi-asserted-by":"publisher","unstructured":"Zhao, J., Li, R., Jin, Q., Wang, X., and Li, H.: MEmoBERT: Pre-training model with prompt-based learning for multimodal emotion recognition. https:\/\/doi.org\/10.48550\/ARXIV.2111.00865 2021","DOI":"10.48550\/ARXIV.2111.00865"},{"key":"1310_CR39","doi-asserted-by":"publisher","first-page":"306","DOI":"10.1016\/j.inffus.2023.02.028","volume":"95","author":"L Zhu","year":"2023","unstructured":"Zhu, L., Zhu, Z., Zhang, C., Xu, Y., Kong, X.: Multimodal sentiment analysis based on fusion methods: a survey. Information Fusion 95, 306\u2013325 (2023). https:\/\/doi.org\/10.1016\/j.inffus.2023.02.028","journal-title":"Information Fusion"},{"key":"1310_CR40","doi-asserted-by":"publisher","unstructured":"Zuo, H., Liu, R., Zhao, J., Gao, G., & Li, H.: Exploiting modality-invariant feature for robust multimodal emotion recognition with missing modalities. ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), 1\u20135. https:\/\/doi.org\/10.1109\/ICASSP49357.2023.10095836 2023","DOI":"10.1109\/ICASSP49357.2023.10095836"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-024-01310-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-024-01310-2\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-024-01310-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,5]],"date-time":"2024-07-05T17:18:22Z","timestamp":1720199902000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-024-01310-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,5,9]]},"references-count":40,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2024,6]]}},"alternative-id":["1310"],"URL":"https:\/\/doi.org\/10.1007\/s00530-024-01310-2","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,5,9]]},"assertion":[{"value":"26 August 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 March 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 May 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.The authors declare the following financial interests\/personal relationships which may be considered as potential competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"144"}}