{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T03:41:33Z","timestamp":1783741293131,"version":"3.55.0"},"reference-count":28,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2024,9,3]],"date-time":"2024-09-03T00:00:00Z","timestamp":1725321600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,9,3]],"date-time":"2024-09-03T00:00:00Z","timestamp":1725321600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["61771299"],"award-info":[{"award-number":["61771299"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Multimed Info Retr"],"published-print":{"date-parts":[[2024,12]]},"DOI":"10.1007\/s13735-024-00347-3","type":"journal-article","created":{"date-parts":[[2024,9,3]],"date-time":"2024-09-03T08:02:16Z","timestamp":1725350536000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":70,"title":["Multi-modal emotion recognition using tensor decomposition fusion and self-supervised multi-tasking"],"prefix":"10.1007","volume":"13","author":[{"given":"Rui","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiawei","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shoujin","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tao","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jingze","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xianxun","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,9,3]]},"reference":[{"key":"347_CR1","volume-title":"Smart cities: IoT technologies, big data solutions, cloud platforms, and cybersecurity techniques","year":"2023","unstructured":"Khang A, Gupta SK, Rani S, Karras DA (eds) (2023) Smart cities: IoT technologies, big data solutions, cloud platforms, and cybersecurity techniques. CRC Press, Boca Raton"},{"key":"347_CR2","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2023.121692","volume":"237","author":"S Zhang","year":"2023","unstructured":"Zhang S, Yang Y, Chen C, Zhang X, Leng Q, Zhao X (2023) Deep learning-based multimodal emotion recognition from audio, visual, and text modalities: a systematic review of recent advancements and future prospects. Expert Syst Appl 237:121692","journal-title":"Expert Syst Appl"},{"key":"347_CR3","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2023.126866","volume":"561","author":"B Pan","year":"2023","unstructured":"Pan B, Hirota K, Jia Z, Dai Y (2023) A review of multimodal emotion recognition from datasets, preprocessing, features, and fusion methods. Neurocomputing 561:126866","journal-title":"Neurocomputing"},{"key":"347_CR4","unstructured":"Liu W, Qiu JL, Zheng WL, Lu BL (2019) Multimodal emotion recognition using deep canonical correlation analysis. arXiv preprint arXiv:1908.05349"},{"issue":"5","key":"347_CR5","doi-asserted-by":"publisher","first-page":"829","DOI":"10.1162\/neco_a_01273","volume":"32","author":"J Gao","year":"2020","unstructured":"Gao J, Li P, Chen Z, Zhang J (2020) A survey on deep learning for multimodal data fusion. Neural Comput 32(5):829\u2013864","journal-title":"Neural Comput"},{"key":"347_CR6","doi-asserted-by":"crossref","unstructured":"Wang Q, Wang J, Quan X, Feng F, Xu Z, Nie S, Wang S, Khabsa M, Firooz H, Liu D (2023) Mustie: Multimodal structural transformer for web information extraction. In proceedings of the 61st annual meeting of the association for computational linguistics (Volume 1: Long Papers) (pp. 2405-2420)","DOI":"10.18653\/v1\/2023.acl-long.135"},{"key":"347_CR7","doi-asserted-by":"crossref","unstructured":"Yang H, Yin L, Zhou Y, Gu J (2021) Exploiting semantic embedding and visual feature for facial action unit detection. In proceedings of the IEEE\/CVF conference on computer vision and pattern recognition(pp. 10482-10491)","DOI":"10.1109\/CVPR46437.2021.01034"},{"key":"347_CR8","doi-asserted-by":"crossref","unstructured":"Yin D, Meng T, Chang KW (2020) Sentibert: A transferable transformer-based architecture for compositional sentiment semantics. arXiv preprint arXiv:2005.04114","DOI":"10.18653\/v1\/2020.acl-main.341"},{"key":"347_CR9","doi-asserted-by":"crossref","unstructured":"Yang K, Xu H, Gao K (2020) Cm-bert: Cross-modal bert for text-audio sentiment analysis. In proceedings of the 28th ACM international conference on multimedia (pp. 521-528)","DOI":"10.1145\/3394171.3413690"},{"key":"347_CR10","doi-asserted-by":"crossref","unstructured":"Park G, Han C, Yoon W, Kim D (2020) MHSAN: multi-head self-attention network for visual semantic embedding. In: proceedings of the IEEE\/CVF winter conference on applications of computer vision (pp. 1518-1526)","DOI":"10.1109\/WACV45572.2020.9093548"},{"key":"347_CR11","doi-asserted-by":"crossref","unstructured":"Kim T, Lee B (2020) Multi-attention multimodal sentiment analysis. In proceedings of the 2020 international conference on multimedia retrieval(pp. 436-441)","DOI":"10.1145\/3372278.3390698"},{"issue":"1","key":"347_CR12","doi-asserted-by":"publisher","DOI":"10.1103\/PhysRevResearch.6.013029","volume":"6","author":"R Levy","year":"2024","unstructured":"Levy R, Luo D, Clark BK (2024) Classical shadows for quantum process tomography on near-term quantum computers. Phys Rev Res 6(1):013029","journal-title":"Phys Rev Res"},{"issue":"7","key":"347_CR13","doi-asserted-by":"publisher","first-page":"3281","DOI":"10.1021\/jacs.9b10780","volume":"142","author":"LO Jones","year":"2020","unstructured":"Jones LO, Mosquera MA, Schatz GC, Ratner MA (2020) Embedding methods for quantum chemistry: applications from materials to life sciences. J Am Chem Soc 142(7):3281\u20133295","journal-title":"J Am Chem Soc"},{"key":"347_CR14","doi-asserted-by":"crossref","unstructured":"Degottex G, Kane J, Drugman T, Raitio T, Scherer S (2014) COVAREP-A collaborative voice analysis repository for speech technologies. In 2014 IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 960-964). IEEE","DOI":"10.1109\/ICASSP.2014.6853739"},{"key":"347_CR15","doi-asserted-by":"crossref","unstructured":"Yuan X, Li L, Wang Y (2019) Nonlinear dynamic soft sensor modeling with supervised long short-term memory network. IEEE Trans Ind Inf 16(5):3168-3176","DOI":"10.1109\/TII.2019.2902129"},{"key":"347_CR16","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2023.111276","volume":"284","author":"J Li","year":"2024","unstructured":"Li J, Zhang X, Li F, Duan S, Huang L (2024) Acoustic-articulatory emotion recognition using multiple features and parameter-optimized cascaded deep learning network. Know Based Syst 284:111276","journal-title":"Know Based Syst"},{"key":"347_CR17","doi-asserted-by":"crossref","unstructured":"Lee C, Kim S, Han D, Yang H, Park YW, Kwon BC, Ko S (2020) GUIComp: A GUI design assistant with real-time, multi-faceted feedback. In proceedings of the 2020 CHI conference on human factors in computing systems (pp. 1-13)","DOI":"10.1145\/3313831.3376327"},{"key":"347_CR18","doi-asserted-by":"crossref","unstructured":"Zadeh A, Chen M, Poria S, Cambria E, Morency LP (2017) Tensor fusion network for multimodal sentiment analysis. arXiv preprint arXiv:1707.07250","DOI":"10.18653\/v1\/D17-1115"},{"key":"347_CR19","unstructured":"Malik OA, Becker S (2018) Low-rank tucker decomposition of large tensors using tensorsketch. Advances in Neural Information Processing Systems, 31"},{"key":"347_CR20","doi-asserted-by":"crossref","unstructured":"Tellamekala MK, Amiriparian S, Schuller BW, Andr\u00e9 E, Giesbrecht T, Valstar M (2023) COLD fusion: Calibrated and ordinal latent distribution fusion for uncertainty-aware multimodal emotion recognition. IEEE Transactions on Pattern Analysis and Machine Intelligence","DOI":"10.1109\/TPAMI.2023.3325770"},{"key":"347_CR21","unstructured":"AmirZadeh Rowan Zellers, Eli Pincus (2016) Louis-Philippe Morency. MOSI: Multimodal corpus of sentiment intensity and subjectivity analysis in online opinion videos. CoRR, abs\/1606.06259"},{"key":"347_CR22","unstructured":"Amir Zadeh (2018) CMU-MOSEI dataset. http:\/\/multicomp.cs.cmu.edu\/resources\/cmu-mosei-dataset\/,. Accessed: 2018"},{"key":"347_CR23","doi-asserted-by":"crossref","unstructured":"Zadeh A, Chen M, Poria S, Cambria E, Morency LP (2017) Tensor fusion network for multimodal sentiment analysis. arXiv preprint arXiv:1707.07250","DOI":"10.18653\/v1\/D17-1115"},{"key":"347_CR24","doi-asserted-by":"crossref","unstructured":"Liu Z, Shen Y, Lakshminarasimhan VB, Liang PP, Zadeh A, Morency LP (2018) Efficient low-rank multimodal fusion with modality-specific factors. arXiv preprint arXiv:1806.00064","DOI":"10.18653\/v1\/P18-1209"},{"key":"347_CR25","unstructured":"Tsai YHH, Liang PP, Zadeh A, Morency LP, Salakhutdinov R (2018) Learning factorized multimodal representations. arXiv preprint arXiv:1806.06176"},{"key":"347_CR26","doi-asserted-by":"crossref","unstructured":"Hazarika D, Zimmermann R, Poria S (2020) Misa: Modality-invariant and-specific representations for multimodal sentiment analysis. In proceedings of the 28th ACM international conference on multimedia (pp. 1122-1131)","DOI":"10.1145\/3394171.3413678"},{"key":"347_CR27","doi-asserted-by":"crossref","unstructured":"Yu W, Xu H, Yuan Z, Wu J (2021) Learning modality-specific representations with self-supervised multi-task learning for multimodal sentiment analysis. In proceedings of the AAAI conference on artificial intelligence (Vol. 35, No. 12, pp. 10790-10797)","DOI":"10.1609\/aaai.v35i12.17289"},{"key":"347_CR28","doi-asserted-by":"crossref","unstructured":"Rahman W, Hasan MK, Lee S, Zadeh A, Mao C, Morency LP, Hoque E (2020) Integrating multimodal information in large pretrained transformers. In proceedings of the conference. Association for computational linguistics. Meeting (Vol. 2020, p. 2359). NIH Public Access","DOI":"10.18653\/v1\/2020.acl-main.214"}],"container-title":["International Journal of Multimedia Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-024-00347-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13735-024-00347-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-024-00347-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,28]],"date-time":"2024-11-28T04:08:07Z","timestamp":1732766887000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13735-024-00347-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,9,3]]},"references-count":28,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2024,12]]}},"alternative-id":["347"],"URL":"https:\/\/doi.org\/10.1007\/s13735-024-00347-3","relation":{"has-preprint":[{"id-type":"doi","id":"10.21203\/rs.3.rs-3916468\/v1","asserted-by":"object"}]},"ISSN":["2192-6611","2192-662X"],"issn-type":[{"value":"2192-6611","type":"print"},{"value":"2192-662X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,9,3]]},"assertion":[{"value":"1 February 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 June 2024","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 August 2024","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 September 2024","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"There is no conflict of interest in our work.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"39"}}