{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,13]],"date-time":"2026-08-13T08:00:07Z","timestamp":1786608007103,"version":"3.56.0"},"reference-count":239,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2024,6,1]],"date-time":"2024-06-01T00:00:00Z","timestamp":1717200000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,6,1]],"date-time":"2024-06-01T00:00:00Z","timestamp":1717200000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61771299"],"award-info":[{"award-number":["61771299"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Cogn Comput"],"published-print":{"date-parts":[[2024,7]]},"DOI":"10.1007\/s12559-024-10287-z","type":"journal-article","created":{"date-parts":[[2024,6,1]],"date-time":"2024-06-01T14:01:36Z","timestamp":1717250496000},"page":"1504-1530","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":91,"title":["A Review of Key Technologies for Emotion Analysis Using\u00a0Multimodal Information"],"prefix":"10.1007","volume":"16","author":[{"given":"Xianxun","family":"Zhu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chaopeng","family":"Guo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Heyang","family":"Feng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yao","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yichen","family":"Feng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiangyang","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7974-9510","authenticated-orcid":false,"given":"Rui","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,6,1]]},"reference":[{"issue":"1","key":"10287_CR1","doi-asserted-by":"publisher","first-page":"20","DOI":"10.1037\/0033-2909.99.1.20","volume":"99","author":"EB Foa","year":"1986","unstructured":"Foa EB, Kozak MJ. Emotional processing of fear: exposure to corrective information[J]. Psychol Bull. 1986;99(1):20.","journal-title":"Psychol Bull."},{"key":"10287_CR2","doi-asserted-by":"crossref","unstructured":"Ernst H, Scherpf M, Pannasch S, et al. Assessment of the human response to acute mental stress-An overview and a multimodal study[J]. PLoS ONE. 2023;18(11): e0294069.","DOI":"10.1371\/journal.pone.0294069"},{"key":"10287_CR3","doi-asserted-by":"crossref","unstructured":"Liu EH, Chambers CR, Moore C. Fifty years of research on leader communication: What we know and where we are going[J]. The Leadership Quarterly. 2023:101734.","DOI":"10.1016\/j.leaqua.2023.101734"},{"issue":"1","key":"10287_CR4","doi-asserted-by":"publisher","first-page":"145","DOI":"10.1037\/0033-295X.110.1.145","volume":"110","author":"JA Russell","year":"2003","unstructured":"Russell JA. Core affect and the psychological construction of emotion[J]. Psychol Rev. 2003;110(1):145.","journal-title":"Psychol Rev."},{"issue":"02","key":"10287_CR5","first-page":"52","volume":"2","author":"SMSA Abdullah","year":"2021","unstructured":"Abdullah SMSA, Ameen SYA, Sadeeq MAM, et al. Multimodal emotion recognition using deep learning[J]. J Appl Sci Technol Trends. 2021;2(02):52\u20138.","journal-title":"J Appl Sci Technol Trends."},{"key":"10287_CR6","doi-asserted-by":"publisher","first-page":"307","DOI":"10.1007\/978-3-030-16272-6_11","volume":"11400","author":"C Marechal","year":"2019","unstructured":"Marechal C, Mikolajewski D, Tyburek K, et al. Survey on AI-Based Multimodal Methods for Emotion Detection[J]. High-performance modelling and simulation for big data applications. 2019;11400:307\u201324.","journal-title":"High-performance modelling and simulation for big data applications."},{"key":"10287_CR7","doi-asserted-by":"publisher","first-page":"102447","DOI":"10.1016\/j.jnca.2019.102447","volume":"149","author":"NJ Shoumy","year":"2020","unstructured":"Shoumy NJ, Ang LM, Seng KP, et al. Multimodal big data affective analytics: A comprehensive survey using text, audio, visual and physiological signals[J]. J Netw Comput Appl. 2020;149:102447.","journal-title":"J Netw Comput Appl."},{"issue":"10","key":"10287_CR8","doi-asserted-by":"publisher","first-page":"6729","DOI":"10.1109\/TPAMI.2021.3094362","volume":"44","author":"S Zhao","year":"2021","unstructured":"Zhao S, Yao X, Yang J, et al. Affective image content analysis: Two decades review and new perspectives[J]. IEEE Trans Pattern Anal Mach Intell. 2021;44(10):6729\u201351.","journal-title":"IEEE Trans Pattern Anal Mach Intell."},{"issue":"1","key":"10287_CR9","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1186\/s40537-021-00459-1","volume":"8","author":"H Christian","year":"2021","unstructured":"Christian H, Suhartono D, Chowanda A, et al. Text based personality prediction from multiple social media data sources using pre-trained language model and model averaging[J]. J Big Data. 2021;8(1):1\u201320.","journal-title":"J Big Data."},{"key":"10287_CR10","doi-asserted-by":"crossref","unstructured":"Das R, Singh T D. Multimodal Sentiment Analysis: A Survey of Methods, Trends and Challenges[J]. ACM Comput Surv. 2023.","DOI":"10.1145\/3586075"},{"key":"10287_CR11","doi-asserted-by":"crossref","unstructured":"Zhu L, Zhu Z, Zhang C, et\u00a0al. Multimodal sentiment analysis based on fusion methods: A survey[J]. Inform Fusion. 2023.","DOI":"10.1016\/j.inffus.2023.02.028"},{"key":"10287_CR12","volume":"17","author":"N Ahmed","year":"2023","unstructured":"Ahmed N, Al Aghbari Z, Girija S. A systematic survey on multimodal emotion recognition using learning algorithms[J]. Intell Syst Appl. 2023;17: 200171.","journal-title":"Intell Syst Appl."},{"issue":"2s","key":"10287_CR13","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3545572","volume":"19","author":"S Jabeen","year":"2023","unstructured":"Jabeen S, Li X, Amin MS, et al. A Review on Methods and Applications in Multimodal Deep Learning[J]. ACM Trans Multimed Comput Commun Appl. 2023;19(2s):1\u201341.","journal-title":"ACM Trans Multimed Comput Commun Appl."},{"key":"10287_CR14","doi-asserted-by":"crossref","unstructured":"Gandhi A, Adhvaryu K, Poria S, et\u00a0al. Multimodal sentiment analysis: A systematic review of history, datasets, multimodal fusion methods, applications, challenges and future directions[J]. Inform Fusion. 2022.","DOI":"10.1016\/j.inffus.2022.09.025"},{"issue":"11","key":"10287_CR15","doi-asserted-by":"publisher","first-page":"163","DOI":"10.3390\/computers11110163","volume":"11","author":"GM Dimitri","year":"2022","unstructured":"Dimitri GM. A Short Survey on Deep Learning for Multimodal Integration: Applications, Future Perspectives and Challenges[J]. Computers. 2022;11(11):163.","journal-title":"Computers."},{"issue":"7","key":"10287_CR16","first-page":"1479","volume":"16","author":"Z Xiaoming","year":"2022","unstructured":"Xiaoming Z, Yijiao Y, Shiqing Z. Survey of Deep Learning Based Multimodal Emotion Recognition[J]. J Front Comput Sci Technol. 2022;16(7):1479.","journal-title":"J Front Comput Sci Technol."},{"issue":"1","key":"10287_CR17","doi-asserted-by":"publisher","first-page":"327","DOI":"10.3390\/app12010327","volume":"12","author":"C Luna-Jimenez","year":"2021","unstructured":"Luna-Jimenez C, Kleinlein R, Griol D, et al. A proposal for multimodal emotion recognition using aural transformers and action units on RAVDESS dataset[J]. Appl Sci. 2021;12(1):327.","journal-title":"Appl Sci."},{"issue":"5","key":"10287_CR18","volume":"11","author":"G Chandrasekaran","year":"2021","unstructured":"Chandrasekaran G, Nguyen TN, Hemanth DJ. Multimodal sentimental analysis for social media applications: A comprehensive review[J]. Wiley Interdisciplinary Reviews: Data Mining and Knowledge Discovery. 2021;11(5): e1415.","journal-title":"Wiley Interdisciplinary Reviews: Data Mining and Knowledge Discovery."},{"issue":"6","key":"10287_CR19","doi-asserted-by":"publisher","first-page":"59","DOI":"10.1109\/MSP.2021.3106895","volume":"38","author":"S Zhao","year":"2021","unstructured":"Zhao S, Jia G, Yang J, et al. Emotion recognition from multiple modalities: Fundamentals and methodologies[J]. IEEE Signal Process Mag. 2021;38(6):59\u201373.","journal-title":"IEEE Signal Process Mag."},{"key":"10287_CR20","doi-asserted-by":"publisher","first-page":"204","DOI":"10.1016\/j.inffus.2021.06.003","volume":"76","author":"SA Abdu","year":"2021","unstructured":"Abdu SA, Yousef AH, Salem A. Multimodal video sentiment analysis using deep learning approaches, a survey[J]. Inform Fusion. 2021;76:204\u201326.","journal-title":"Inform Fusion."},{"key":"10287_CR21","doi-asserted-by":"crossref","unstructured":"Sharma G, Dhall A. A survey on automatic multimodal emotion recognition in the wild[J]. Advances in data science: Methodol Appl. 2021:35-64.","DOI":"10.1007\/978-3-030-51870-7_3"},{"key":"10287_CR22","doi-asserted-by":"crossref","unstructured":"Nandi A, Xhafa F, Subirats L, et\u00a0al. A survey on multimodal data stream mining for e-learner\u2019s emotion recognition[C]. In: 2020 International Conference on Omni-layer Intelligent Systems (COINS). IEEE; 2020. p. 1\u20136.","DOI":"10.1109\/COINS49042.2020.9191370"},{"key":"10287_CR23","doi-asserted-by":"publisher","first-page":"103","DOI":"10.1016\/j.inffus.2020.01.011","volume":"59","author":"J Zhang","year":"2020","unstructured":"Zhang J, Yin Z, Chen P, et al. Emotion recognition using multi-modal data and machine learning techniques: A tutorial and review[J]. Inform Fusion. 2020;59:103\u201326.","journal-title":"Inform Fusion."},{"key":"10287_CR24","doi-asserted-by":"publisher","first-page":"90982","DOI":"10.1109\/ACCESS.2019.2926751","volume":"7","author":"JKP Seng","year":"2019","unstructured":"Seng JKP, Ang KLM. Multimodal emotion and sentiment modeling from unstructured Big data: Challenges, architecture, and techniques[J]. IEEE Access. 2019;7:90982\u201398.","journal-title":"IEEE Access."},{"key":"10287_CR25","doi-asserted-by":"crossref","unstructured":"Baltru?aitis T, Ahuja C, Morency LP. Multimodal machine learning: A survey and taxonomy[J]. IEEE Trans Pattern Anal Mach Intell. 2018;41(2):423\u201343.","DOI":"10.1109\/TPAMI.2018.2798607"},{"key":"10287_CR26","doi-asserted-by":"publisher","first-page":"98","DOI":"10.1016\/j.inffus.2017.02.003","volume":"37","author":"S Poria","year":"2017","unstructured":"Poria S, Cambria E, Bajpai R, et al. A review of affective computing: From unimodal analysis to multimodal fusion[J]. Inform Fusion. 2017;37:98\u2013125.","journal-title":"Inform Fusion."},{"key":"10287_CR27","doi-asserted-by":"crossref","unstructured":"Latha CP, Priya M. A review on deep learning algorithms for speech and facial emotion recognition[J]. APTIKOM J Comput Sci Inf Technol. 2016;1(3):92\u2013108.","DOI":"10.11591\/APTIKOM.J.CSIT.118"},{"key":"10287_CR28","doi-asserted-by":"crossref","unstructured":"Schuller B, Valstar M, Eyben F, et\u00a0al. Avec 2011-the first international audio\/visual emotion challenge[C]. Affective Computing and Intelligent Interaction: Fourth International Conference, ACII 2011, Memphis, TN, USA, October 9-12, 2011, Proceedings, Part II. Springer Berlin Heidelberg, 2011:415-424.","DOI":"10.1007\/978-3-642-24571-8_53"},{"key":"10287_CR29","doi-asserted-by":"crossref","unstructured":"Schuller B, Valstar M, Eyben F, McKeown G, Cowie R, Pantic M. Avec 2011-the first international audio\/visual emotion challenge. In Affective Computing and Intelligent Interaction, 2011, p. 415-424. Springer Berlin Heidelberg.","DOI":"10.1007\/978-3-642-24571-8_53"},{"key":"10287_CR30","doi-asserted-by":"crossref","unstructured":"Chen H, Zhou H, Du J, et\u00a0al. The first multimodal information based speech processing challenge:Data, tasks, baselines and results. In Processing ICASSP. 2022, p. 9266-9270. IEEE.","DOI":"10.1109\/ICASSP43922.2022.9746683"},{"key":"10287_CR31","doi-asserted-by":"crossref","unstructured":"Zafeiriou S, Kollias D, Nicolaou M A, et\u00a0al. Aff-wild: valence and arousal\u2019In-the-Wild\u2019challenge[C]. Proceedings of the IEEE conference on computer vision and pattern recognition workshops. 2017:34-41.","DOI":"10.1109\/CVPRW.2017.248"},{"issue":"1","key":"10287_CR32","doi-asserted-by":"publisher","first-page":"43","DOI":"10.1109\/TAFFC.2015.2396531","volume":"6","author":"Y Baveye","year":"2015","unstructured":"Baveye Y, Dellandrea E, Chamaret C, et al. LIRIS-ACCEDE: A video database for affective content analysis[J]. IEEE Trans Affect Comput. 2015;6(1):43\u201355.","journal-title":"IEEE Trans Affect Comput."},{"key":"10287_CR33","doi-asserted-by":"crossref","unstructured":"Stappen L, Baird A, Rizos G, et\u00a0al. Muse 2020 challenge and workshop: Multimodal sentiment analysis, emotion-target engagement and trustworthiness detection in real-life media: Emotional car reviews in-the-wild[C]. Proceedings of the 1st International on Multimodal Sentiment Analysis in Real-life Media Challenge and Workshop. 2020:35-44.","DOI":"10.1145\/3423327.3423673"},{"key":"10287_CR34","doi-asserted-by":"crossref","unstructured":"Li Y, Tao J, Schuller B, et\u00a0al. Mec 2017: Multimodal emotion recognition challenge[C]. 2018 First Asian Conference on Affective Computing and Intelligent Interaction (ACII Asia). IEEE, 2018:1-5.","DOI":"10.1109\/ACIIAsia.2018.8470342"},{"key":"10287_CR35","doi-asserted-by":"crossref","unstructured":"Kollias D. Abaw: valence-arousal estimation, expression recognition, action unit detection & multi-task learning challenges[C]. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 2022:2328-2336.","DOI":"10.1109\/CVPRW56347.2022.00259"},{"key":"10287_CR36","doi-asserted-by":"crossref","unstructured":"Lian Z, Sun H, Sun L, et\u00a0al. Mer 2023: Multi-label learning, modality robustness, and semi-supervised learning[C]. In: Proceedings of the 31st ACM International Conference on Multimedia. 2023:9610-9614.","DOI":"10.1145\/3581783.3612836"},{"key":"10287_CR37","doi-asserted-by":"crossref","unstructured":"Li J, Zhang Z, Lang J, et\u00a0al. Hybrid multimodal feature extraction, mining and fusion for sentiment analysis[C]. In: Proceedings of the 3rd International on Multimodal Sentiment Analysis Workshop and Challenge. 2022:81-88.","DOI":"10.1145\/3551876.3554809"},{"key":"10287_CR38","doi-asserted-by":"crossref","unstructured":"Zong D, Ding C, Li B, et\u00a0al. Building robust multimodal sentiment recognition via a simple yet effective multimodal transformer[C]. In: Proceedings of the 31st ACM International Conference on Multimedia. 2023:9596-9600.","DOI":"10.1145\/3581783.3612872"},{"key":"10287_CR39","unstructured":"Advances in Neural Information Processing Systems 10: Proceedings of the 1997 Conference[M]. Mit Press, 1998."},{"key":"10287_CR40","unstructured":"Amsaleg L, Huet B, Larson M, et\u00a0al. Proceedings of the 27th ACM International Conference on Multimedia[C]. 27th ACM International Conference on Multimedia. ACM Press, 2019."},{"key":"10287_CR41","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2021.103635","volume":"303","author":"V Lomonaco","year":"2022","unstructured":"Lomonaco V, Pellegrini L, Rodriguez P, et al. Cvpr 2020 continual learning in computer vision competition: Approaches, results, current challenges and future directions[J]. Artif Intell. 2022;303: 103635.","journal-title":"Artif Intell."},{"key":"10287_CR42","doi-asserted-by":"crossref","unstructured":"Gatterbauer W, Kumar A. Guest Editors\u2019 Introduction to the Special Section on the 33rd International Conference on Data Engineering (ICDE 2017)[J]. IEEE Trans Knowl Data Eng. 2019;31(7):1222-1223.","DOI":"10.1109\/TKDE.2019.2912043"},{"key":"10287_CR43","unstructured":"Liu Y, Paek T, Patwardhan M. Proceedings of the 2018 Conference of the North American Chapter of the Association for Computational Linguistics: Demonstrations[C]. Proceedings of the 2018 Conference of the North American Chapter of the Association for Computational Linguistics: Demonstrations. 2018."},{"key":"10287_CR44","unstructured":"Lang J. Proceedings of the Twenty-Seventh International Joint Conference on Artificial Intelligence (IJCAI 2018)[J]. 2018."},{"key":"10287_CR45","doi-asserted-by":"crossref","unstructured":"Reddy C K A, Dubey H, Gopal V, et\u00a0al. ICASSP 2021 deep noise suppression challenge[C]. ICASSP 2021-2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 2021:6623-6627.","DOI":"10.1109\/ICASSP39728.2021.9415105"},{"key":"10287_CR46","unstructured":"Morency L P, Bohus D, Aghajan H, et\u00a0al. ICMI\u201912: Proceedings of the ACM SIGCHI 14th International Conference on Multimodal Interaction[C]. 14th International Conference on Multimodal Interaction, ICMI 2012. Association for Computing Machinery (ACM), 2012."},{"key":"10287_CR47","first-page":"692","volume":"2022","author":"N Nitta","year":"2022","unstructured":"Nitta N, Hu A, Tobitani K. MMArt-ACM 2022: 5th Joint Workshop on Multimedia Artworks Analysis and Attractiveness Computing in Multimedia[C]. Proceedings of the International Conference on Multimedia Retrieval. 2022;2022:692\u20133.","journal-title":"Proceedings of the International Conference on Multimedia Retrieval."},{"key":"10287_CR48","unstructured":"PRICAI 2022: Trends in Artificial Intelligence: 19th Pacific Rim International Conference on Artificial Intelligence, PRICAI 2022, Shanghai, China, November 10-13, 2022, Proceedings, Part III[M]. Springer Nature, 2022."},{"key":"10287_CR49","unstructured":"Gabbouj M. Proceedings of WIAMIS 2001: Workshop on Image Analysis for Multimedia Services[J]. 2001."},{"issue":"2","key":"10287_CR50","doi-asserted-by":"publisher","first-page":"179","DOI":"10.1097\/01.psy.0000155663.93160.d2","volume":"67","author":"PC Strike","year":"2005","unstructured":"Strike PC, Steptoe A. Behavioral and emotional triggers of acute coronary syndromes: a systematic review and critique[J]. Psychosom Med. 2005;67(2):179\u201386.","journal-title":"Psychosom Med."},{"issue":"2","key":"10287_CR51","doi-asserted-by":"publisher","first-page":"131","DOI":"10.1016\/0167-8760(91)90005-I","volume":"11","author":"W Hubert","year":"1991","unstructured":"Hubert W, de Jong-Meyer R. Autonomic, neuroendocrine, and subjective responses to emotion-inducing film stimuli[J]. Int J Psychophysiol. 1991;11(2):131\u201340.","journal-title":"Int J Psychophysiol."},{"key":"10287_CR52","doi-asserted-by":"crossref","unstructured":"Bhattacharyya MR, Steptoe A. Emotional triggers of acute coronary syndromes: strength of evidence, biological processes, and clinical implications[J]. Prog Cardiovasc Dis. 2007;49(5):353\u201365.","DOI":"10.1016\/j.pcad.2006.11.002"},{"issue":"12","key":"10287_CR53","doi-asserted-by":"publisher","first-page":"1030","DOI":"10.3390\/ani9121030","volume":"9","author":"C Scopa","year":"2019","unstructured":"Scopa C, Contalbrigo L, Greco A, et al. Emotional transfer in human-horse interaction: New perspectives on equine assisted interventions[J]. Animals. 2019;9(12):1030.","journal-title":"Animals."},{"issue":"21","key":"10287_CR54","doi-asserted-by":"publisher","first-page":"5824","DOI":"10.1039\/D0BM01284J","volume":"8","author":"JK Hong","year":"2020","unstructured":"Hong JK, Gao L, Singh J, et al. Evaluating medical device and material thrombosis under flow: current and emerging technologies[J]. Biomater Sci. 2020;8(21):5824\u201345.","journal-title":"Biomater Sci."},{"issue":"2","key":"10287_CR55","doi-asserted-by":"publisher","first-page":"209","DOI":"10.1016\/j.ijpsycho.2004.07.006","volume":"55","author":"K Werheid","year":"2005","unstructured":"Werheid K, Alpay G, Jentzsch I, et al. Priming emotional facial expressions as evidenced by event-related brain potentials[J]. Int J Psychophysiol. 2005;55(2):209\u201319.","journal-title":"Int J Psychophysiol."},{"issue":"4","key":"10287_CR56","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1037\/0022-3514.87.4.529","volume":"87","author":"D Matsumoto","year":"2004","unstructured":"Matsumoto D, Ekman P. The relationship among expressions, labels, and descriptions of contempt[J]. J Pers Soc Psychol. 2004;87(4):529.","journal-title":"J Pers Soc Psychol."},{"key":"10287_CR57","unstructured":"Picard R W. Affective computing[M]. MIT press, 2000."},{"key":"10287_CR58","unstructured":"Tomkins S S. Affect imagery consciousness: the complete edition: two volumes[M]. Springer publishing company, 2008."},{"key":"10287_CR59","doi-asserted-by":"publisher","first-page":"331","DOI":"10.1007\/BF02229025","volume":"19","author":"A Mehrabian","year":"1997","unstructured":"Mehrabian A. Comparison of the PAD and PANAS as models for describing emotions and for differentiating anxiety from depression[J]. J Psychopathol Behav Assess. 1997;19:331\u201357.","journal-title":"J Psychopathol Behav Assess."},{"issue":"1","key":"10287_CR60","doi-asserted-by":"publisher","first-page":"145","DOI":"10.1037\/0033-295X.110.1.145","volume":"110","author":"JA Russell","year":"2003","unstructured":"Russell JA. Core affect and the psychological construction of emotion[J]. Psychol Rev. 2003;110(1):145.","journal-title":"Psychol Rev."},{"issue":"3","key":"10287_CR61","doi-asserted-by":"publisher","first-page":"715","DOI":"10.1017\/S0954579405050340","volume":"17","author":"J Posner","year":"2005","unstructured":"Posner J, Russell JA, Peterson BS. The circumplex model of affect: An integrative approach to affective neuroscience, cognitive development, and psychopathology[J]. Dev Psychopathol. 2005;17(3):715\u201334.","journal-title":"Dev Psychopathol."},{"issue":"2","key":"10287_CR62","doi-asserted-by":"publisher","first-page":"180","DOI":"10.1016\/j.jamcollsurg.2009.04.010","volume":"209","author":"RJ Bleicher","year":"2009","unstructured":"Bleicher RJ, Ciocca RM, Egleston BL, et al. Association of routine pretreatment magnetic resonance imaging with time to surgery, mastectomy rate, and margin status[J]. J Am Coll Surg. 2009;209(2):180\u20137.","journal-title":"J Am Coll Surg."},{"key":"10287_CR63","doi-asserted-by":"crossref","unstructured":"Swathi C, Anoop B K, Dhas D A S, et\u00a0al. Comparison of different image preprocessing methods used for retinal fundus images[C]. 2017 Conference on Emerging Devices and Smart Systems (ICEDSS). IEEE, 2017:175-179.","DOI":"10.1109\/ICEDSS.2017.8073677"},{"key":"10287_CR64","doi-asserted-by":"crossref","unstructured":"Finlayson G D, Schiele B, Crowley J L. Comprehensive colour image normalization[C]. Computer Vision-ECCV\u201998: 5th European Conference on Computer Vision Freiburg, Germany, June, 2-6, 1998 Proceedings, Volume I 5. Springer Berlin Heidelberg, 1998:475-490.","DOI":"10.1007\/BFb0055685"},{"issue":"1","key":"10287_CR65","first-page":"39","volume":"3","author":"AK Vishwakarma","year":"2012","unstructured":"Vishwakarma AK, Mishra A. Color image enhancement techniques: a critical review[J]. Indian J Comput Sci Eng. 2012;3(1):39\u201345.","journal-title":"Indian J Comput Sci Eng."},{"issue":"10","key":"10287_CR66","doi-asserted-by":"publisher","first-page":"3810","DOI":"10.1016\/j.patcog.2012.03.019","volume":"45","author":"T Celik","year":"2012","unstructured":"Celik T. Two-dimensional histogram equalization and contrast enhancement[J]. Pattern Recogn. 2012;45(10):3810\u201324.","journal-title":"Pattern Recogn."},{"key":"10287_CR67","doi-asserted-by":"crossref","unstructured":"Jayaram S, Schmugge S, Shin M C, et\u00a0al. Effect of colorspace transformation, the illuminance component, and color modeling on skin detection[C]. Proceedings of the 2004 IEEE Computer Society Conference on Computer Vision and Pattern Recognition, 2004. CVPR 2004. IEEE, 2004, 2:II-II.","DOI":"10.1109\/CVPR.2004.1315248"},{"key":"10287_CR68","doi-asserted-by":"crossref","unstructured":"Pandey M, Bhatia M, Bansal A, An anatomization of noise removal techniques on medical images[C]. international conference on innovation and challenges in cyber security (iciccs-inbush). IEEE. 2016;2016:224\u20139.","DOI":"10.1109\/ICICCS.2016.7542308"},{"issue":"1","key":"10287_CR69","first-page":"1","volume":"3","author":"R Maini","year":"2009","unstructured":"Maini R, Aggarwal H. Study and comparison of various image edge detection techniques[J]. Int J Image Process (IJIP). 2009;3(1):1\u201311.","journal-title":"Int J Image Process (IJIP)"},{"key":"10287_CR70","doi-asserted-by":"crossref","unstructured":"Eltanany AS, SAfy Elwan M, Amein AS. Key point detection techniques[C]. Proceedings of the International Conference on Advanced Intelligent Systems and Informatics 2019. Springer International Publishing. 2020:901-911.","DOI":"10.1007\/978-3-030-31129-2_82"},{"key":"10287_CR71","doi-asserted-by":"crossref","unstructured":"Yang MH, Kriegman DJ, Ahuja N. Detecting faces in images: a survey[J]. IEEE Trans Pattern Anal Mach Intell. 2002;24(1):34\u201358.","DOI":"10.1109\/34.982883"},{"key":"10287_CR72","unstructured":"Qin J, He ZS. ASVM, face recognition method based on Gabor-featured key points[C]. international conference on machine learning and cybernetics. IEEE. 2005;2005(8):5144\u20139."},{"key":"10287_CR73","doi-asserted-by":"crossref","unstructured":"Xiong X, De la Torre F. Supervised descent method and its applications to face alignment[C]. Proceedings of the IEEE conference on computer vision and pattern recognition. 2013:532-539.","DOI":"10.1109\/CVPR.2013.75"},{"issue":"1","key":"10287_CR74","doi-asserted-by":"publisher","first-page":"126","DOI":"10.1037\/0022-0663.92.1.126","volume":"92","author":"S Kalyuga","year":"2000","unstructured":"Kalyuga S, Chandler P, Sweller J. Incorporating learner experience into the design of multimedia instruction[J]. J Educ Psychol. 2000;92(1):126.","journal-title":"J Educ Psychol."},{"key":"10287_CR75","doi-asserted-by":"crossref","unstructured":"Bezoui M, Elmoutaouakkil A, Beni-hssane A. Feature extraction of some Quranic recitation using mel-frequency cepstral coeficients (MFCC)[C]. 5th international conference on multimedia computing and systems (ICMCS). IEEE. 2016;2016:127\u201331.","DOI":"10.1109\/ICMCS.2016.7905619"},{"key":"10287_CR76","unstructured":"Shrawankar U, Thakare V M. Adverse conditions and ASR techniques for robust speech user interface[J]. arXiv preprint arXiv:1303.5515, 2013."},{"key":"10287_CR77","doi-asserted-by":"crossref","unstructured":"Liu L, He J, Palm G. Signal modeling for speaker identification. In: Proceedings of the 1996 IEEE International Conference on Acoustics, Speech, and Signal Processing (vol. 2). IEEE; 1996. pp. 665\u20138.","DOI":"10.1109\/ICASSP.1996.543208"},{"issue":"3","key":"10287_CR78","doi-asserted-by":"publisher","first-page":"159","DOI":"10.1016\/j.specom.2006.12.004","volume":"49","author":"B Bozkurt","year":"2007","unstructured":"Bozkurt B, Couvreur L, Dutoit T. Chirp group delay analysis of speech signals[J]. Speech Commun. 2007;49(3):159\u201376.","journal-title":"Speech Commun."},{"issue":"3","key":"10287_CR79","first-page":"1628","volume":"2010","author":"N Seman","year":"2010","unstructured":"Seman N, Bakar ZA, Bakar NA. An evaluation of endpoint detection measures for Malay speech recognition of an isolated words[C]. International Symposium on Information Technology, IEEE. 2010;2010(3):1628\u201335.","journal-title":"International Symposium on Information Technology, IEEE."},{"key":"10287_CR80","doi-asserted-by":"crossref","unstructured":"Hua Y, Guo J, Zhao H. Deep belief networks and deep learning[C]. Proceedings of 2015 International Conference on Intelligent Computing and Internet of Things, IEEE. 2015:1-4.","DOI":"10.1109\/ICAIOT.2015.7111524"},{"issue":"3","key":"10287_CR81","doi-asserted-by":"publisher","first-page":"822","DOI":"10.3758\/BRM.40.3.822","volume":"40","author":"MJ Owren","year":"2008","unstructured":"Owren MJ. GSU Praat Tools: scripts for modifying and analyzing sounds using Praat acoustics software[J]. Behav Res Methods. 2008;40(3):822\u20139.","journal-title":"Behav Res Methods."},{"key":"10287_CR82","doi-asserted-by":"crossref","unstructured":"Eyben F, Wllmer M, Schuller B. Opensmile: the munich versatile and fast open-source audio feature extractor[C]. Proceedings of the 18th ACM international conference on Multimedia. 2010:1459-1462.","DOI":"10.1145\/1873951.1874246"},{"key":"10287_CR83","doi-asserted-by":"crossref","unstructured":"Hossan M A, Memon S, Gregory M A. A novel approach for MFCC feature extraction[C]. In: 2010 4th International Conference on Signal Processing and Communication Systems. IEEE, 2010:1-5.","DOI":"10.1109\/ICSPCS.2010.5709752"},{"key":"10287_CR84","doi-asserted-by":"crossref","unstructured":"Acheampong F A, Nunoo-Mensah H, Chen W. Transformer models for text-based emotion detection: a review of BERT-based approaches[J]. Artificial Intelligence Review, 2021:1-41.","DOI":"10.1007\/s10462-021-09958-2"},{"key":"10287_CR85","doi-asserted-by":"crossref","unstructured":"Mishra B, Fernandes SL, Abhishek K, et al. Facial expression recognition using feature based techniques and model based techniques: a survey[C]. In: 2nd international conference on electronics and communication systems (ICECS), IEEE. 2015;2015:589\u201394.","DOI":"10.1109\/ECS.2015.7124976"},{"key":"10287_CR86","doi-asserted-by":"crossref","unstructured":"Mastropaolo A, Scalabrino S, Cooper N, et\u00a0al. Studying the usage of text-to-text transfer transformer to support code-related tasks[C]. In: 2021 IEEE\/ACM 43rd International Conference on Software Engineering (ICSE). IEEE, 2021:336-347.","DOI":"10.1109\/ICSE43902.2021.00041"},{"key":"10287_CR87","unstructured":"Qian F, Han J. Contrastive regularization for multimodal emotion recognition using audio and text[J]. arXiv preprint arXiv:2211.10885, 2022."},{"key":"10287_CR88","doi-asserted-by":"crossref","unstructured":"Zhang Y, Wang J, Liu Y, et\u00a0al. A Multitask learning model for multimodal sarcasm, sentiment and emotion recognition in conversations[J]. Inform Fusion. 2023.","DOI":"10.1016\/j.inffus.2023.01.005"},{"issue":"9","key":"10287_CR89","doi-asserted-by":"publisher","first-page":"13617","DOI":"10.1007\/s11042-022-13762-7","volume":"82","author":"C Fuente","year":"2023","unstructured":"Fuente C, Castellanos FJ, Valero-Mas JJ, et al. Multimodal recognition of frustration during game-play with deep neural networks[J]. Multimed Tools Appl. 2023;82(9):13617\u201336.","journal-title":"Multimed Tools Appl."},{"key":"10287_CR90","doi-asserted-by":"crossref","unstructured":"Li J, Wang X, Lv G, et\u00a0al. GA2MIF: graph and attention based two-stage multi-source Information Fusion for Conversational Emotion Detection[J]. IEEE Trans Affect Comput. 2023.","DOI":"10.1109\/TAFFC.2023.3261279"},{"key":"10287_CR91","doi-asserted-by":"crossref","unstructured":"Wang B, Dong G, Zhao Y, et\u00a0al. Hierarchically stacked graph convolution for emotion recognition in conversation[J]. Knowledge-Based Systems, 2023:110285.","DOI":"10.1016\/j.knosys.2023.110285"},{"key":"10287_CR92","doi-asserted-by":"crossref","unstructured":"Padi S, Sadjadi S O, Manocha D, et\u00a0al. Multimodal emotion recognition using transfer learning from speaker recognition and Bert-based models[J]. arXiv preprint arXiv:2202.08974, 2022.","DOI":"10.21437\/Odyssey.2022-57"},{"key":"10287_CR93","doi-asserted-by":"crossref","unstructured":"Tran D, Bourdev L, Fergus R, et\u00a0al. Learning spatiotemporal features with 3d convolutional networks[C]. In: Proceedings of the IEEE international conference on computer vision. 2015:4489-4497.","DOI":"10.1109\/ICCV.2015.510"},{"key":"10287_CR94","unstructured":"Bansal K, Agarwal H, Joshi A, et\u00a0al. Shapes of emotions: multimodal emotion recognition in conversations via emotion shifts[C]. In: Proceedings of the First Workshop on Performance and Interpretability Evaluations of Multimodal, Multipurpose, Massive-Scale Models. 2022:44-56."},{"key":"10287_CR95","doi-asserted-by":"crossref","unstructured":"Tang S, Luo Z, Nan G, et al. Fusion with hierarchical graphs for multimodal emotion recognition[C]. In: Asia-Pacific Signal and Information Processing Association Annual Summit and Conference (APSIPA ASC), IEEE. 2022;2022:1288\u201396.","DOI":"10.23919\/APSIPAASC55919.2022.9979932"},{"key":"10287_CR96","unstructured":"Qian F, Han J. Contrastive regularization for multimodal emotion recognition using audio and text[J]. arXiv preprint arXiv:2211.10885, 2022."},{"key":"10287_CR97","doi-asserted-by":"crossref","unstructured":"Wei Q, Huang X, Zhang Y. FV2ES: a fully end2end multimodal system for fast yet effective video emotion recognition inference[J]. IEEE Transactions on Broadcasting, 2022.","DOI":"10.1109\/TBC.2022.3215245"},{"key":"10287_CR98","doi-asserted-by":"crossref","unstructured":"Wu Y, Li J. Multi-modal emotion identification fusing facial expression and EEG[J]. Multimed Tools Appl. 2023;82(7):10901\u201319.","DOI":"10.1007\/s11042-022-13711-4"},{"issue":"1","key":"10287_CR99","doi-asserted-by":"publisher","DOI":"10.1111\/jsr.13634","volume":"32","author":"MJ Reid","year":"2023","unstructured":"Reid MJ, Omlin X, Espie CA, et al. The effect of sleep continuity disruption on multimodal emotion processing and regulation: a laboratory based, randomised, controlled experiment in good sleepers[J]. J Sleep Res. 2023;32(1): e13634.","journal-title":"J Sleep Res."},{"key":"10287_CR100","doi-asserted-by":"publisher","DOI":"10.1016\/j.bspc.2022.104561","volume":"82","author":"M Fang","year":"2023","unstructured":"Fang M, Peng S, Liang Y, et al. A multimodal fusion model with multi-level attention mechanism for depression detection[J]. Biomed Signal Process Control. 2023;82: 104561.","journal-title":"Biomed Signal Process Control."},{"key":"10287_CR101","doi-asserted-by":"crossref","unstructured":"Stappen L, Baird A, Rizos G, et\u00a0al. Muse 2020 challenge and workshop: Multimodal sentiment analysis, emotion-target engagement and trustworthiness detection in real-life media: emotional car reviews in-the-wild[C]. In: Proceedings of the 1st International on Multimodal Sentiment Analysis in Real-life Media Challenge and Workshop. 2020:35-44.","DOI":"10.1145\/3423327.3423673"},{"key":"10287_CR102","unstructured":"Miranda J A, Canabal M F, Portela Garca M, et\u00a0al. Embedded emotion recognition: autonomous multimodal affective internet of things[C]. In: Proceedings of the cyber-physical systems workshop. 2018, 2208:22-29."},{"key":"10287_CR103","doi-asserted-by":"crossref","unstructured":"Caesar H, Bankiti V, Lang A H, et\u00a0al. nuscenes: a multimodal dataset for autonomous driving[C]. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 2020:11621-11631.","DOI":"10.1109\/CVPR42600.2020.01164"},{"key":"10287_CR104","doi-asserted-by":"crossref","unstructured":"Mangano G, Ferrari A, Rafele C, et\u00a0al. Willingness of sharing facial data for emotion recognition: a case study in the insurance market[J]. AI & SOCIETY. 2023:1-12..","DOI":"10.1007\/s00146-023-01690-5"},{"issue":"CSCW1","key":"10287_CR105","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3579528","volume":"7","author":"KL Boyd","year":"2023","unstructured":"Boyd KL, Andalibi N. Automated emotion recognition in the workplace: How proposed technologies reveal potential futures of work[J]. Proceedings of the ACM on Human-Computer Interaction. 2023;7(CSCW1):1\u201337.","journal-title":"Proceedings of the ACM on Human-Computer Interaction."},{"key":"10287_CR106","doi-asserted-by":"publisher","first-page":"1272","DOI":"10.22214\/ijraset.2023.49225","volume":"11","author":"A Dubey","year":"2023","unstructured":"Dubey A, Shingala B, Panara JR, et al. Digital content recommendation system through facial emotion recognition[J]. Int J Res Appl Sci Eng Technol. 2023;11:1272\u20136.","journal-title":"Int J Res Appl Sci Eng Technol."},{"key":"10287_CR107","doi-asserted-by":"crossref","unstructured":"Holding B C, Laukka P, Fischer H, et\u00a0al. Multimodal emotion recognition is resilient to insufficient sleep: results from cross-sectional and experimental studies[J]. Sleep. 2017;40(11):zsx145.","DOI":"10.1093\/sleep\/zsx145"},{"key":"10287_CR108","doi-asserted-by":"publisher","first-page":"35","DOI":"10.1016\/j.entcs.2019.04.009","volume":"343","author":"M Egger","year":"2019","unstructured":"Egger M, Ley M, Hanke S. Emotion recognition from physiological signal analysis: a review[J]. Electron Notes Theor Comput Sci. 2019;343:35\u201355.","journal-title":"Electron Notes Theor Comput Sci."},{"issue":"3","key":"10287_CR109","doi-asserted-by":"publisher","first-page":"304","DOI":"10.1037\/neu0000323","volume":"31","author":"SC Andrews","year":"2017","unstructured":"Andrews SC, Staios M, Howe J, et al. Multimodal emotion processing deficits are present in amyotrophic lateral sclerosis[J]. Neuropsychology. 2017;31(3):304.","journal-title":"Neuropsychology."},{"key":"10287_CR110","unstructured":"O\u2019Shea K, Nash R. An introduction to convolutional neural networks[J]. arXiv preprint arXiv:1511.08458, 2015."},{"key":"10287_CR111","unstructured":"Meignier S, Merlin T. LIUM SpkDiarization: an open source toolkit for diarization[C]. CMU SPUD Workshop. 2010."},{"key":"10287_CR112","unstructured":"Povey D, Ghoshal A, Boulianne G, et\u00a0al. The Kaldi speech recognition toolkit[C]. IEEE 2011 workshop on automatic speech recognition and understanding. IEEE Signal Processing Society, 2011 (CONF)."},{"key":"10287_CR113","unstructured":"Gaida C, Lange P, Petrick R, et\u00a0al. Comparing open-source speech recognition toolkits[C]. 11th International Workshop on Natural Language Processing and Cognitive Science. 2014."},{"key":"10287_CR114","unstructured":"Moffat D, Ronan D, Reiss J D. An evaluation of audio feature extraction toolboxes[J]. 2015."},{"key":"10287_CR115","doi-asserted-by":"crossref","unstructured":"Karkada D, Saletore VA. Training speech recognition models on HPC infrastructure[C]. IEEE\/ACM Machine Learning in HPC Environments (MLHPC), IEEE. 2018;2018:124\u201332.","DOI":"10.1109\/MLHPC.2018.8638637"},{"key":"10287_CR116","doi-asserted-by":"crossref","unstructured":"Syed M S S, Stolar M, Pirogova E, et\u00a0al. Speech acoustic features characterising individuals with high and low public trust[C]. 2019 13th International Conference on Signal Processing and Communication Systems (ICSPCS). IEEE, 2019:1-9.","DOI":"10.1109\/ICSPCS47537.2019.9008747"},{"key":"10287_CR117","doi-asserted-by":"crossref","unstructured":"Degottex G, Kane J, Drugman T, et al. COVAREP-a collaborative voice analysis repository for speech technologies[C]. In: IEEE international conference on acoustics, speech and signal processing (icassp), IEEE. 2014;2014:960\u20134.","DOI":"10.1109\/ICASSP.2014.6853739"},{"issue":"1","key":"10287_CR118","doi-asserted-by":"publisher","DOI":"10.1002\/cpe.7407","volume":"35","author":"U Yadav","year":"2023","unstructured":"Yadav U, Sharma AK, Patil D. Review of automated depression detection: social posts, audio and video, open challenges and future direction[J]. Concurrency and Computation: Practice and Experience. 2023;35(1): e7407.","journal-title":"Concurrency and Computation: Practice and Experience."},{"issue":"1","key":"10287_CR119","first-page":"7","volume":"5","author":"S Vijayarani","year":"2015","unstructured":"Vijayarani S, Ilamathi MJ, Nithya M. Preprocessing techniques for text mining-an overview[J]. International Journal of Computer Science and Communication Networks. 2015;5(1):7\u201316.","journal-title":"International Journal of Computer Science and Communication Networks."},{"issue":"12","key":"10287_CR120","doi-asserted-by":"publisher","first-page":"2544","DOI":"10.1002\/asi.21416","volume":"61","author":"M Thelwall","year":"2010","unstructured":"Thelwall M, Buckley K, Paltoglou G, et al. Sentiment strength detection in short informal text[J]. J Am Soc Inform Sci Technol. 2010;61(12):2544\u201358.","journal-title":"J Am Soc Inform Sci Technol."},{"key":"10287_CR121","doi-asserted-by":"crossref","unstructured":"Wu Z, King S. Investigating gated recurrent networks for speech synthesis[C]. In: 2016 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 2016:5140-5144.","DOI":"10.1109\/ICASSP.2016.7472657"},{"issue":"1","key":"10287_CR122","doi-asserted-by":"publisher","first-page":"93","DOI":"10.1038\/s41746-021-00464-x","volume":"4","author":"DM Korngiebel","year":"2021","unstructured":"Korngiebel DM, Mooney SD. Considering the possibilities and pitfalls of Generative Pre-trained Transformer 3 (GPT-3) in healthcare delivery[J]. NPJ Digital Medicine. 2021;4(1):93.","journal-title":"NPJ Digital Medicine."},{"key":"10287_CR123","unstructured":"Liu Y, Ott M, Goyal N, et\u00a0al. Roberta: a robustly optimized bert pretraining approach[J]. arXiv preprint arXiv:1907.11692, 2019."},{"key":"10287_CR124","doi-asserted-by":"crossref","unstructured":"Zahidi Y, El Younoussi Y, Al-Amrani Y. Different valuable tools for Arabic sentiment analysis: a comparative evaluation[J]. International Journal of Electrical and Computer Engineering (2088-8708), 2021, 11(1).","DOI":"10.11591\/ijece.v11i1.pp753-762"},{"key":"10287_CR125","doi-asserted-by":"publisher","DOI":"10.1016\/j.bspc.2023.105661","volume":"89","author":"H Cai","year":"2024","unstructured":"Cai H, Lin Q, Liu H, et al. Recognition of human mood, alertness and comfort under the influence of indoor lighting using physiological features[J]. Biomed Signal Process Control. 2024;89: 105661.","journal-title":"Biomed Signal Process Control."},{"key":"10287_CR126","doi-asserted-by":"publisher","DOI":"10.1016\/j.jecp.2023.105757","volume":"237","author":"E Tan","year":"2024","unstructured":"Tan E, Hamlin JK. Toddlers\u2019 affective responses to sociomoral scenes: Insights from physiological measures[J]. J Exp Child Psychol. 2024;237: 105757.","journal-title":"J Exp Child Psychol."},{"issue":"1","key":"10287_CR127","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0296468","volume":"19","author":"M Awada","year":"2024","unstructured":"Awada M, Becerik Gerber B, Lucas GM, et al. Stress appraisal in the workplace and its associations with productivity and mood: Insights from a multimodal machine learning analysis[J]. PLoS ONE. 2024;19(1): e0296468.","journal-title":"PLoS ONE."},{"key":"10287_CR128","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2023.111199","volume":"283","author":"W Guo","year":"2024","unstructured":"Guo W, Li Y, Liu M, et al. Functional connectivity-enhanced feature-grouped attention network for cross-subject EEG emotion recognition[J]. Knowl-Based Syst. 2024;283: 111199.","journal-title":"Knowl-Based Syst."},{"issue":"4","key":"10287_CR129","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3616019","volume":"4","author":"EK Naeini","year":"2023","unstructured":"Naeini EK, Sarhaddi F, Azimi I, et al. A deep learning-based PPG quality assessment approach for heart rate and heart rate variability[J]. ACM Transactions on Computing for Healthcare. 2023;4(4):1\u201322.","journal-title":"ACM Transactions on Computing for Healthcare."},{"issue":"8","key":"10287_CR130","doi-asserted-by":"publisher","first-page":"1394","DOI":"10.3390\/medicina59081394","volume":"59","author":"F Panjaitan","year":"2023","unstructured":"Panjaitan F, Nurmaini S, Partan RU. Accurate prediction of sudden cardiac death based on heart rate variability analysis using convolutional neural network[J]. Medicina. 2023;59(8):1394.","journal-title":"Medicina."},{"issue":"1","key":"10287_CR131","doi-asserted-by":"publisher","first-page":"35","DOI":"10.1007\/s10484-022-09558-y","volume":"48","author":"K Nashiro","year":"2023","unstructured":"Nashiro K, Yoo HJ, Cho C, et al. Effects of a randomised trial of 5-week heart rate variability biofeedback intervention on cognitive function: possible benefits for inhibitory control[J]. Appl Psychophysiol Biofeedback. 2023;48(1):35\u201348.","journal-title":"Appl Psychophysiol Biofeedback."},{"key":"10287_CR132","doi-asserted-by":"crossref","unstructured":"Qi N, Piao Y, Yu P, et\u00a0al. Predicting epileptic seizures based on EEG signals using spatial depth features of a 3D-2D hybrid CNN[J]. Medical & Biological Engineering & Computing, 2023:1-12.","DOI":"10.1007\/s11517-023-02792-4"},{"key":"10287_CR133","doi-asserted-by":"publisher","DOI":"10.1016\/j.bspc.2023.105679","volume":"88","author":"D Cho","year":"2024","unstructured":"Cho D, Lee B. Automatic sleep-stage classification based on residual unit and attention networks using directed transfer function of electroencephalogram signals[J]. Biomed Signal Process Control. 2024;88: 105679.","journal-title":"Biomed Signal Process Control."},{"key":"10287_CR134","doi-asserted-by":"crossref","unstructured":"Li Z, Xu B, Zhu C, et\u00a0al. CLMLF: a contrastive learning and multi-layer fusion method for multimodal sentiment detection[J]. arXiv preprint arXiv:2204.05515, 2022.","DOI":"10.18653\/v1\/2022.findings-naacl.175"},{"key":"10287_CR135","doi-asserted-by":"crossref","unstructured":"Yoon S, Byun S, Jung K, Multimodal speech emotion recognition using audio and text[C]. In,. IEEE Spoken Language Technology Workshop (SLT). IEEE. 2018;2018:112\u20138.","DOI":"10.1109\/SLT.2018.8639583"},{"key":"10287_CR136","doi-asserted-by":"crossref","unstructured":"Hazarika D, Poria S, Zadeh A, et\u00a0al. Conversational memory network for emotion recognition in dyadic dialogue videos[C]. In: Proceedings of the conference. Association for Computational Linguistics. North American Chapter. Meeting. NIH Public Access, 2018, 2018:2122.","DOI":"10.18653\/v1\/N18-1193"},{"key":"10287_CR137","doi-asserted-by":"crossref","unstructured":"Mai S, Hu H, Xing S. Divide, conquer and combine: hierarchical feature fusion network with local and global perspectives for multimodal affective computing[C]. In: Proceedings of the 57th annual meeting of the association for computational linguistics. 2019:481-492.","DOI":"10.18653\/v1\/P19-1046"},{"key":"10287_CR138","doi-asserted-by":"crossref","unstructured":"You Q, Luo J, Jin H, et\u00a0al. Cross-modality consistent regression for joint visual-textual sentiment analysis of social multimedia[C]. In: Proceedings of the Ninth ACM international conference on Web search and data mining. 2016:13-22.","DOI":"10.1145\/2835776.2835779"},{"key":"10287_CR139","doi-asserted-by":"crossref","unstructured":"Chen M, Wang S, Liang P P, et\u00a0al. Multimodal sentiment analysis with word-level fusion and reinforcement learning[C]. In: Proceedings of the 19th ACM international conference on multimodal interaction. 2017:163-171.","DOI":"10.1145\/3136755.3136801"},{"key":"10287_CR140","doi-asserted-by":"crossref","unstructured":"Zadeh A, Chen M, Poria S, et\u00a0al. Tensor fusion network for multimodal sentiment analysis[J]. arXiv preprint arXiv:1707.07250, 2017.","DOI":"10.18653\/v1\/D17-1115"},{"key":"10287_CR141","volume-title":"Self-adaptive representation learning model for multi-modal sentiment and sarcasm joint analysis[J]","author":"Y Zhang","year":"2023","unstructured":"Zhang Y, Yu Y, Wang M, et al. Self-adaptive representation learning model for multi-modal sentiment and sarcasm joint analysis[J]. Communications and Applications: ACM Transactions on Multimedia Computing; 2023."},{"key":"10287_CR142","doi-asserted-by":"crossref","unstructured":"Poria S, Cambria E, Hazarika D, et\u00a0al. Context-dependent sentiment analysis in user-generated videos[C]. In: Proceedings of the 55th annual meeting of the association for computational linguistics (volume 1: Long papers). 2017:873-883.","DOI":"10.18653\/v1\/P17-1081"},{"key":"10287_CR143","doi-asserted-by":"crossref","unstructured":"Poria S, Chaturvedi I, Cambria E, et al. Convolutional MKL, based multimodal emotion recognition and sentiment analysis[C]. In: IEEE 16th international conference on data mining (ICDM), IEEE. 2016;2016:439\u201348.","DOI":"10.1109\/ICDM.2016.0055"},{"key":"10287_CR144","unstructured":"Deng D, Zhou Y, Pi J, et\u00a0al. Multimodal utterance-level affect analysis using visual, audio and text features[J]. arXiv preprint arXiv:1805.00625, 2018."},{"key":"10287_CR145","unstructured":"Chen F, Luo Z, Xu Y, et\u00a0al. Complementary fusion of multi-features and multi-modalities in sentiment analysis[J]. arXiv preprint arXiv:1904.08138, 2019."},{"key":"10287_CR146","doi-asserted-by":"crossref","unstructured":"Kumar A, Vepa J. Gated mechanism for attention based multi modal sentiment analysis[C]. In: ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 2020:4477-4481.","DOI":"10.1109\/ICASSP40776.2020.9053012"},{"key":"10287_CR147","doi-asserted-by":"crossref","unstructured":"Xu N, Mao W. Multisentinet: a deep semantic network for multimodal sentiment analysis[C]. In: Proceedings of the. ACM on Conference on Information and Knowledge Management. 2017;2017:2399\u2013402.","DOI":"10.1145\/3132847.3133142"},{"key":"10287_CR148","doi-asserted-by":"publisher","first-page":"429","DOI":"10.1109\/TASLP.2019.2957872","volume":"28","author":"J Yu","year":"2019","unstructured":"Yu J, Jiang J, Xia R. Entity-sensitive attention and fusion network for entity-level multimodal sentiment classification[J]. IEEE\/ACM Transactions on Audio, Speech, and Language Processing. 2019;28:429\u201339.","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing."},{"key":"10287_CR149","doi-asserted-by":"publisher","first-page":"1424","DOI":"10.1109\/TASLP.2021.3068598","volume":"29","author":"S Mai","year":"2021","unstructured":"Mai S, Xing S, Hu H. Analyzing multimodal sentiment via acoustic-and visual-LSTM with channel-aware temporal convolution network[J]. IEEE\/ACM Transactions on Audio, Speech, and Language Processing. 2021;29:1424\u201337.","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing."},{"key":"10287_CR150","doi-asserted-by":"crossref","unstructured":"Xu N, Mao W, Chen G. Multi-interactive memory network for aspect based multimodal sentiment analysis[C]. In: Proceedings of the AAAI Conference on Artificial Intelligence. 2019, 33(01):371-378.","DOI":"10.1609\/aaai.v33i01.3301371"},{"issue":"2","key":"10287_CR151","doi-asserted-by":"publisher","first-page":"22","DOI":"10.1007\/s10723-021-09564-0","volume":"19","author":"D Liu","year":"2021","unstructured":"Liu D, Chen L, Wang Z, et al. Speech expression multimodal emotion recognition based on deep belief network[J]. Journal of Grid Computing. 2021;19(2):22.","journal-title":"Journal of Grid Computing."},{"issue":"1","key":"10287_CR152","doi-asserted-by":"publisher","first-page":"289","DOI":"10.1007\/s12559-022-10073-9","volume":"15","author":"F Wang","year":"2023","unstructured":"Wang F, Tian S, Yu L, et al. TEDT: transformer-based encoding-decoding translation network for multimodal sentiment analysis[J]. Cogn Comput. 2023;15(1):289\u2013303.","journal-title":"Cogn Comput."},{"key":"10287_CR153","doi-asserted-by":"crossref","unstructured":"Kumar A, Vepa J. Gated mechanism for attention based multi modal sentiment analysis[C]. In: ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 2020:4477-4481.","DOI":"10.1109\/ICASSP40776.2020.9053012"},{"key":"10287_CR154","unstructured":"Lu Y, Zheng W, Li B, et\u00a0al. Combining eye movements and EEG to enhance emotion recognition. In: Proceedings of the Twenty-fourth International Joint Conference on Artificial Intelligence, International Joint Conferences on Artificial Intelligence Organization, 2015:1170-1176."},{"issue":"2","key":"10287_CR155","doi-asserted-by":"publisher","first-page":"41","DOI":"10.3390\/a9020041","volume":"9","author":"Y Yu","year":"2016","unstructured":"Yu Y, Lin H, Meng J, et al. Visual and textual sentiment analysis of a microblog using deep convolutional neural networks. Algorithms. 2016;9(2):41.","journal-title":"Algorithms."},{"key":"10287_CR156","doi-asserted-by":"crossref","unstructured":"Poria S, Cambria E, Gelbukh A. Deep convolutional neural network textual features and multiple kernel learning for utterance-level multimodal sentiment analysis. In: Proceedings of the 2015 Conference on Empirical Methods in Natural Language Processing, Association for Computational Linguistics, 2015:2539-2544.","DOI":"10.18653\/v1\/D15-1303"},{"key":"10287_CR157","doi-asserted-by":"crossref","unstructured":"Wang HH, Meghawat A, Morency LP, et\u00a0al. Select-additive learning: improving generalization in multimodal sentiment analysis. In: Proceedings of the 2017 IEEE International Conference on Multimedia and Expo, IEEE Computer Society, 2017:949-954.","DOI":"10.1109\/ICME.2017.8019301"},{"key":"10287_CR158","doi-asserted-by":"crossref","unstructured":"Yu HL, Gui LK, Madaio M, et\u00a0al. Temporally selective attention model for social and affective state recognition in multimedia content. In: Proceedings of the 25th ACM International Conference on Multimedia, ACM, 2017:1743-1751.","DOI":"10.1145\/3123266.3123413"},{"key":"10287_CR159","doi-asserted-by":"crossref","unstructured":"Williams J, Comanescu R, Radu O, et\u00a0al. DNN multimodal fusion techniques for predicting video sentiment. In: Proceedings of Grand Challenge and Workshop on Human Multimodal Language (Challenge-HML), 2018:64-72.","DOI":"10.18653\/v1\/W18-3309"},{"key":"10287_CR160","doi-asserted-by":"crossref","unstructured":"Gkoumas, D., Li, Q., Dehdashti, S., et\u00a0al. Quantum cognitively motivated decision fusion for video sentiment analysis. In: Proceedings of the AAAI Conference on Artificial Intelligence, 2021, 35(1):827-835.","DOI":"10.1609\/aaai.v35i1.16165"},{"key":"10287_CR161","doi-asserted-by":"crossref","unstructured":"Sun, J., Yin, H., Tian, Y., et\u00a0al. Two-level multimodal fusion for sentiment analysis in public security. Security and Communication Networks, 2021.","DOI":"10.1155\/2021\/6662337"},{"key":"10287_CR162","doi-asserted-by":"publisher","first-page":"296","DOI":"10.1016\/j.inffus.2022.07.006","volume":"88","author":"F Zhang","year":"2022","unstructured":"Zhang F, Li XC, Lim CP, et al. Deep emotional arousal network for multimodal sentiment analysis and emotion recognition[J]. Inform Fusion. 2022;88:296\u2013304.","journal-title":"Inform Fusion."},{"key":"10287_CR163","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2022.109259","volume":"136","author":"D Wang","year":"2023","unstructured":"Wang D, Guo X, Tian Y, et al. TETFN: a text enhanced transformer fusion network for multimodal sentiment analysis[J]. Pattern Recogn. 2023;136: 109259.","journal-title":"Pattern Recogn."},{"issue":"3","key":"10287_CR164","doi-asserted-by":"publisher","first-page":"1110","DOI":"10.1109\/TCYB.2018.2797176","volume":"49","author":"W Zheng","year":"2018","unstructured":"Zheng W, Liu W, Lu Y, et al. Emotionmeter: a multimodal framework for recognizing human emotions. IEEE Transactions on Cybernetics. 2018;49(3):1110\u201322.","journal-title":"IEEE Transactions on Cybernetics."},{"issue":"10","key":"10287_CR165","first-page":"1","volume":"28","author":"S Zhang","year":"2017","unstructured":"Zhang S, Zhang S, Huang T, et al. Learning affective features with a hybrid deep model for audio-visual emotion recognition. IEEE Trans Circuits Syst Video Technol. 2017;28(10):1\u20131.","journal-title":"IEEE Trans Circuits Syst Video Technol."},{"key":"10287_CR166","doi-asserted-by":"crossref","unstructured":"Chen M, Wang S, Liang P P, et\u00a0al. Multimodal sentiment analysis with word-level fusion and reinforcement learning[C]. In: Proceedings of the 19th ACM international conference on multimodal interaction. 2017:163-171.","DOI":"10.1145\/3136755.3136801"},{"key":"10287_CR167","doi-asserted-by":"crossref","unstructured":"Shenoy A, Sardana A. Multilogue-net: a context aware RNN for multi-modal emotion detection and sentiment analysis in conversation[J]. arXiv preprint arXiv:2002.08267, 2020.","DOI":"10.18653\/v1\/2020.challengehml-1.3"},{"key":"10287_CR168","doi-asserted-by":"publisher","first-page":"168865","DOI":"10.1109\/ACCESS.2020.3023871","volume":"8","author":"Y Cimtay","year":"2020","unstructured":"Cimtay Y, Ekmekcioglu E, Caglar-Ozhan S. Cross-subject multimodal emotion recognition based on hybrid fusion[J]. IEEE Access. 2020;8:168865\u201378.","journal-title":"IEEE Access."},{"issue":"4","key":"10287_CR169","doi-asserted-by":"publisher","first-page":"1334","DOI":"10.1016\/j.jnca.2006.09.007","volume":"30","author":"H Gunes","year":"2007","unstructured":"Gunes H, Piccardi M. Bi-modal emotion recognition from expressive face and body gestures[J]. J Netw Comput Appl. 2007;30(4):1334\u201345.","journal-title":"J Netw Comput Appl."},{"key":"10287_CR170","doi-asserted-by":"crossref","unstructured":"Paraskevopoulos G, Georgiou E, Potamianos A. Mmlatch: bottom-up top-down fusion for multimodal sentiment analysis[C]. In: ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 2022:4573-4577.","DOI":"10.1109\/ICASSP43922.2022.9746418"},{"key":"10287_CR171","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2023.121363","volume":"236","author":"L Qu","year":"2024","unstructured":"Qu L, Liu S, Wang M, et al. Trans2Fuse: empowering image fusion through self-supervised learning and multi-modal transformations via transformer networks[J]. Expert Syst Appl. 2024;236: 121363.","journal-title":"Expert Syst Appl."},{"key":"10287_CR172","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2023.102161","volume":"104","author":"H Fan","year":"2024","unstructured":"Fan H, Zhang X, Xu Y, et al. Transformer-based multimodal feature enhancement networks for multimodal depression detection integrating video, audio and remote photoplethysmograph signals[J]. Inform Fusion. 2024;104: 102161.","journal-title":"Inform Fusion."},{"key":"10287_CR173","doi-asserted-by":"crossref","unstructured":"Zhu X, Huang Y, Wang X, et\u00a0al. Emotion recognition based on brain-like multimodal hierarchical perception[J]. Multimed Tools Appl. 2023:1-19.","DOI":"10.1007\/s11042-023-17347-w"},{"key":"10287_CR174","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2023.126992","volume":"565","author":"J Huang","year":"2024","unstructured":"Huang J, Pu Y, Zhou D, et al. Dynamic hypergraph convolutional network for multimodal sentiment analysis[J]. Neurocomputing. 2024;565: 126992.","journal-title":"Neurocomputing."},{"key":"10287_CR175","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2023.102039","volume":"102","author":"X Wang","year":"2024","unstructured":"Wang X, Guan Z, Qian W, et al. CS2Fusion: contrastive learning for self-supervised infrared and visible image fusion by estimating feature compensation map[J]. Inform Fusion. 2024;102: 102039.","journal-title":"Inform Fusion."},{"key":"10287_CR176","doi-asserted-by":"publisher","DOI":"10.1016\/j.bspc.2023.105301","volume":"86","author":"Y Han","year":"2023","unstructured":"Han Y, Nie R, Cao J, et al. IE-CFRN: information exchange-based collaborative feature representation network for multi-modal medical image fusion[J]. Biomed Signal Process Control. 2023;86: 105301.","journal-title":"Biomed Signal Process Control."},{"key":"10287_CR177","unstructured":"Ni J, Bai Y, Zhang W, et\u00a0al. Deep equilibrium multimodal fusion[J]. arXiv preprint arXiv:2306.16645, 2023."},{"key":"10287_CR178","doi-asserted-by":"publisher","first-page":"26","DOI":"10.1016\/j.inffus.2023.02.011","volume":"95","author":"H Li","year":"2023","unstructured":"Li H, Zhao J, Li J, et al. Feature dynamic alignment and refinement for infrared-visible image fusion: translation robust fusion[J]. Inform Fusion. 2023;95:26\u201341.","journal-title":"Inform Fusion."},{"key":"10287_CR179","doi-asserted-by":"publisher","DOI":"10.1016\/j.jbi.2023.104466","volume":"145","author":"J Liu","year":"2023","unstructured":"Liu J, Capurro D, Nguyen A, et al. Attention-based multimodal fusion with contrast for robust clinical prediction in the face of missing modalities[J]. J Biomed Inform. 2023;145: 104466.","journal-title":"J Biomed Inform."},{"key":"10287_CR180","unstructured":"Zhang X, Wei X, Zhou Z, et\u00a0al. Dynamic alignment and fusion of multimodal physiological patterns for stress recognition[J]. IEEE Trans Affect Comput. 2023"},{"key":"10287_CR181","doi-asserted-by":"publisher","first-page":"282","DOI":"10.1016\/j.inffus.2023.01.005","volume":"93","author":"Y Zhang","year":"2023","unstructured":"Zhang Y, Wang J, Liu Y, et al. A multitask learning model for multimodal sarcasm, sentiment and emotion recognition in conversations[J]. Inform Fusion. 2023;93:282\u2013301.","journal-title":"Inform Fusion."},{"key":"10287_CR182","doi-asserted-by":"crossref","unstructured":"Liu Y, Zhang X, Kauttonen J, et\u00a0al. Uncertain facial expression recognition via multi-task assisted correction[J]. IEEE Trans Multimed. 2023.","DOI":"10.1109\/TMM.2023.3301209"},{"key":"10287_CR183","doi-asserted-by":"crossref","unstructured":"Liu J, Lin R, Wu G, et\u00a0al. Coconet: coupled contrastive learning network with multi-level feature ensemble for multi-modality image fusion[J]. Int J Comput Vis. 2023:1-28.","DOI":"10.1007\/s11263-023-01952-1"},{"key":"10287_CR184","doi-asserted-by":"crossref","unstructured":"Liu K, Xue F, Guo D, et\u00a0al. Multimodal graph contrastive learning for multimedia-based recommendation[J]. IEEE Trans Multimed. 2023.","DOI":"10.1109\/TMM.2023.3251108"},{"issue":"18","key":"10287_CR185","doi-asserted-by":"publisher","first-page":"3909","DOI":"10.3390\/electronics12183909","volume":"12","author":"J Song","year":"2023","unstructured":"Song J, Chen H, Li C, et al. MIFM: multimodal information fusion model for educational exercises[J]. Electronics. 2023;12(18):3909.","journal-title":"Electronics."},{"key":"10287_CR186","doi-asserted-by":"crossref","unstructured":"Zhang S, Yang Y, Chen C, et\u00a0al. Deep learning-based multimodal emotion recognition from audio, visual, and text modalities: a systematic review of recent advancements and future prospects[J]. Expert Syst Appl. 2023:121692.","DOI":"10.1016\/j.eswa.2023.121692"},{"issue":"34","key":"10287_CR187","doi-asserted-by":"publisher","first-page":"24435","DOI":"10.1007\/s00521-023-09036-4","volume":"35","author":"G Dogan","year":"2023","unstructured":"Dogan G, Akbulut FP. Multi-modal fusion learning through biosignal, audio, and visual content for detection of mental stress[J]. Neural Comput Appl. 2023;35(34):24435\u201354.","journal-title":"Neural Comput Appl."},{"key":"10287_CR188","unstructured":"Liu W, Zuo Y. Stone needle: a general multimodal large-scale model framework towards healthcare[J]. arXiv preprint arXiv:2306.16034, 2023."},{"key":"10287_CR189","doi-asserted-by":"crossref","unstructured":"Zhao X, Li M, Weber C, et\u00a0al. Chat with the environment: interactive multimodal perception using large language models[J]. arXiv preprint arXiv:2303.08268, 2023.","DOI":"10.1109\/IROS55552.2023.10342363"},{"key":"10287_CR190","doi-asserted-by":"publisher","first-page":"37","DOI":"10.1016\/j.inffus.2022.11.022","volume":"92","author":"K Kim","year":"2023","unstructured":"Kim K, Park S. AOBERT: all-modalities-in-one BERT for multimodal sentiment analysis[J]. Inform Fusion. 2023;92:37\u201345.","journal-title":"Inform Fusion."},{"key":"10287_CR191","doi-asserted-by":"crossref","unstructured":"Tong Z, Du N, Song X, et\u00a0al. Study on mindspore deep learning framework[C]. In: 2021 17th International Conference on Computational Intelligence and Security (CIS). IEEE, 2021:183-186.","DOI":"10.1109\/CIS54983.2021.00046"},{"key":"10287_CR192","doi-asserted-by":"crossref","unstructured":"Rasley J, Rajbhandari S, Ruwase O, et\u00a0al. Deepspeed: system optimizations enable training deep learning models with over 100 billion parameters[C]. In: Proceedings of the 26th ACM SIGKDD International Conference on Knowledge Discovery & Data Mining. 2020:3505-3506.","DOI":"10.1145\/3394486.3406703"},{"key":"10287_CR193","doi-asserted-by":"crossref","unstructured":"Huang J, Wang H, Sun Y, et\u00a0al. ERNIE-GeoL: a geography-and-language pre-trained model and its applications in Baidu maps[C]. In: Proceedings of the 28th ACM SIGKDD Conference on Knowledge Discovery and Data Mining. 2022:3029-3039.","DOI":"10.1145\/3534678.3539021"},{"key":"10287_CR194","doi-asserted-by":"publisher","first-page":"335","DOI":"10.1007\/s10579-008-9076-6","volume":"42","author":"C Busso","year":"2008","unstructured":"Busso C, Bulut M, Lee CC, et al. IEMOCAP: interactive emotional dyadic motion capture database[J]. Lang Resour Eval. 2008;42:335\u201359.","journal-title":"Lang Resour Eval."},{"key":"10287_CR195","unstructured":"Zadeh A, Zellers R, Pincus E, et\u00a0al. Mosi: multimodal corpus of sentiment intensity and subjectivity analysis in online opinion videos[J]. arXiv preprint arXiv:1606.06259, 2016."},{"key":"10287_CR196","doi-asserted-by":"crossref","unstructured":"Poria S, Hazarika D, Majumder N, et\u00a0al. Meld: a multimodal multi-party dataset for emotion recognition in conversations[J]. arXiv preprint arXiv:1810.02508, 2018.","DOI":"10.18653\/v1\/P19-1050"},{"key":"10287_CR197","unstructured":"Zadeh A A B, Liang P P, Poria S, et\u00a0al. Multimodal language analysis in the wild: Cmu-mosei dataset and interpretable dynamic fusion graph[C]. In: Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers). 2018:2236-2246."},{"key":"10287_CR198","doi-asserted-by":"crossref","unstructured":"Yu W, Xu H, Meng F, et\u00a0al. Ch-sims: a Chinese multimodal sentiment analysis dataset with fine-grained annotation of modality[C]. In: Proceedings of the 58th annual meeting of the association for computational linguistics. 2020:3718-3727.","DOI":"10.18653\/v1\/2020.acl-main.343"},{"key":"10287_CR199","doi-asserted-by":"crossref","unstructured":"Zafeiriou S, Kollias D, Nicolaou M A, et\u00a0al. Aff-wild: valence and arousal\u2019In-the-Wild\u2019challenge[C]. In: Proceedings of the IEEE conference on computer vision and pattern recognition workshops. 2017:34-41.","DOI":"10.1109\/CVPRW.2017.248"},{"issue":"5","key":"10287_CR200","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0196391","volume":"13","author":"SR Livingstone","year":"2018","unstructured":"Livingstone SR, Russo FA. The Ryerson Audio-Visual Database of Emotional Speech and Song (RAVDESS): a dynamic, multimodal set of facial and vocal expressions in North American English[J]. PLoS ONE. 2018;13(5): e0196391.","journal-title":"PLoS ONE."},{"issue":"1","key":"10287_CR201","doi-asserted-by":"publisher","first-page":"5","DOI":"10.1109\/T-AFFC.2011.20","volume":"3","author":"G McKeown","year":"2011","unstructured":"McKeown G, Valstar M, Cowie R, et al. The semaine database: annotated multimodal records of emotionally colored conversations between a person and a limited agent[J]. IEEE Trans Affect Comput. 2011;3(1):5\u201317.","journal-title":"IEEE Trans Affect Comput."},{"key":"10287_CR202","doi-asserted-by":"publisher","first-page":"8669","DOI":"10.1007\/s00521-020-05616-w","volume":"33","author":"J Chen","year":"2021","unstructured":"Chen J, Wang C, Wang K, et al. HEU Emotion: a large-scale database for multimodal emotion recognition in the wild[J]. Neural Comput Appl. 2021;33:8669\u201385.","journal-title":"Neural Comput Appl."},{"key":"10287_CR203","doi-asserted-by":"crossref","unstructured":"Shen G, Wang X, Duan X, et\u00a0al. Memor: a dataset for multimodal emotion reasoning in videos[C]. In: Proceedings of the 28th ACM International Conference on Multimedia. 2020:493-502.","DOI":"10.1145\/3394171.3413909"},{"issue":"1","key":"10287_CR204","doi-asserted-by":"publisher","DOI":"10.1088\/1741-2552\/ac49a7","volume":"19","author":"X Wu","year":"2022","unstructured":"Wu X, Zheng WL, Li Z, et al. Investigating EEG-based functional connectivity patterns for multimodal emotion recognition[J]. J Neural Eng. 2022;19(1): 016012.","journal-title":"J Neural Eng."},{"key":"10287_CR205","doi-asserted-by":"crossref","unstructured":"Zadeh A, Liang P P, Poria S, et\u00a0al. Multi-attention recurrent network for human communication comprehension[C]. In: Proceedings of the AAAI Conference on Artificial Intelligence. 2018, 32(1).","DOI":"10.1609\/aaai.v32i1.12024"},{"key":"10287_CR206","doi-asserted-by":"crossref","unstructured":"Zadeh A, Liang P P, Mazumder N, et\u00a0al. Memory fusion network for multi-view sequential learning[C]. In: Proceedings of the AAAI conference on artificial intelligence. 2018, 32(1).","DOI":"10.1609\/aaai.v32i1.12021"},{"key":"10287_CR207","doi-asserted-by":"publisher","first-page":"679","DOI":"10.1016\/j.ins.2022.11.076","volume":"619","author":"S Liu","year":"2023","unstructured":"Liu S, Gao P, Li Y, et al. Multi-modal fusion network with complementarity and importance for emotion recognition[J]. Inf Sci. 2023;619:679\u201394.","journal-title":"Inf Sci."},{"key":"10287_CR208","doi-asserted-by":"crossref","unstructured":"Chen F, Shao J, Zhu S, et\u00a0al. Multivariate, multi-frequency and multimodal: rethinking graph neural networks for emotion recognition in conversation[C]. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 2023:10761-10770.","DOI":"10.1109\/CVPR52729.2023.01036"},{"key":"10287_CR209","doi-asserted-by":"crossref","unstructured":"Khan M, Gueaieb W, El Saddik A, et\u00a0al. MSER: multimodal speech emotion recognition using cross-attention with deep fusion[J]. Expert Syst Appl. 2023:122946.","DOI":"10.1016\/j.eswa.2023.122946"},{"key":"10287_CR210","doi-asserted-by":"crossref","unstructured":"Pan J, Fang W, Zhang Z, et\u00a0al. Multimodal emotion recognition based on facial expressions, speech, and EEG[J]. IEEE Open Journal of Engineering in Medicine and Biology, 2023.","DOI":"10.1109\/OJEMB.2023.3240280"},{"key":"10287_CR211","unstructured":"Meng T, Shou Y, Ai W, et\u00a0al. Deep imbalanced learning for multimodal emotion recognition in conversations[J]. arXiv preprint arXiv:2312.06337, 2023."},{"issue":"4","key":"10287_CR212","doi-asserted-by":"publisher","DOI":"10.1007\/s11704-023-2444-y","volume":"18","author":"Z Fu","year":"2024","unstructured":"Fu Z, Liu F, Xu Q, et al. LMR-CBT: learning modality-fused representations with CB-transformer for multimodal emotion recognition from unaligned multimodal sequences[J]. Front Comp Sci. 2024;18(4): 184314.","journal-title":"Front Comp Sci."},{"key":"10287_CR213","doi-asserted-by":"crossref","unstructured":"Ma H, Wang J, Lin H, et\u00a0al. A transformer-based model with self-distillation for multimodal emotion recognition in conversations[J]. IEEE Trans Multimed. 2023.","DOI":"10.1109\/TMM.2023.3271019"},{"key":"10287_CR214","doi-asserted-by":"crossref","unstructured":"Shi T, Huang S L. MultiEMO: an attention-based correlation-aware multimodal fusion framework for emotion recognition in conversations[C]. In: Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers). 2023:14752-14766.","DOI":"10.18653\/v1\/2023.acl-long.824"},{"key":"10287_CR215","unstructured":"Li X. TACOformer: token-channel compounded cross attention for multimodal emotion recognition[J]. arXiv preprint arXiv:2306.13592, 2023."},{"key":"10287_CR216","doi-asserted-by":"crossref","unstructured":"Li J, Wang X, Lv G, et\u00a0al. Graphcfc: a directed graph based cross-modal feature complementation approach for multimodal conversational emotion recognition[J]. IEEE Trans Multimed. 2023.","DOI":"10.1109\/TMM.2023.3260635"},{"key":"10287_CR217","doi-asserted-by":"crossref","unstructured":"Palash M, Bhargava B. EMERSK\u2013explainable multimodal emotion recognition with situational knowledge[J]. arXiv preprint arXiv:2306.08657, 2023.","DOI":"10.1109\/TMM.2023.3304015"},{"key":"10287_CR218","doi-asserted-by":"crossref","unstructured":"Li Y, Wang Y, Cui Z. Decoupled multimodal distilling for emotion recognition[C]. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 2023:6631-6640.","DOI":"10.1109\/CVPR52729.2023.00641"},{"key":"10287_CR219","doi-asserted-by":"publisher","first-page":"14742","DOI":"10.1109\/ACCESS.2023.3244390","volume":"11","author":"HD Le","year":"2023","unstructured":"Le HD, Lee GS, Kim SH, et al. Multi-label multimodal emotion recognition with transformer-based fusion and emotion-level representation learning[J]. IEEE Access. 2023;11:14742\u201351.","journal-title":"IEEE Access."},{"key":"10287_CR220","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2023.102129","volume":"103","author":"J Tang","year":"2024","unstructured":"Tang J, Ma Z, Gan K, et al. Hierarchical multimodal-fusion of physiological signals for emotion recognition with scenario adaption and contrastive alignment[J]. Inform Fusion. 2024;103: 102129.","journal-title":"Inform Fusion."},{"issue":"4","key":"10287_CR221","doi-asserted-by":"publisher","first-page":"1834","DOI":"10.3390\/s23041834","volume":"23","author":"Y He","year":"2023","unstructured":"He Y, Seng KP, Ang LM. multimodal sensor-input architecture with deep learning for audio-visual speech recognition in wild[J]. Sensors. 2023;23(4):1834.","journal-title":"Sensors."},{"key":"10287_CR222","doi-asserted-by":"crossref","unstructured":"Stappen L, Schumann L, Sertolli B, et\u00a0al. Muse-toolbox: the multimodal sentiment analysis continuous annotation fusion and discrete class transformation toolbox[M]. In: Proceedings of the 2nd on Multimodal Sentiment Analysis Challenge. 2021:75-82.","DOI":"10.1145\/3475957.3484451"},{"key":"10287_CR223","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2023.102129","volume":"103","author":"J Tang","year":"2024","unstructured":"Tang J, Ma Z, Gan K, et al. Hierarchical multimodal-fusion of physiological signals for emotion recognition with scenario adaption and contrastive alignment[J]. Inform Fusion. 2024;103: 102129.","journal-title":"Inform Fusion."},{"key":"10287_CR224","unstructured":"Wang W, Arora R, Livescu K, et\u00a0al. On deep multi-view representation learning[C]. In: International conference on machine learning. PMLR, 2015:1083-1092."},{"issue":"4","key":"10287_CR225","doi-asserted-by":"publisher","first-page":"1250","DOI":"10.1109\/TNNLS.2018.2856253","volume":"30","author":"Y Yu","year":"2018","unstructured":"Yu Y, Tang S, Aizawa K, et al. Category-based deep CCA for fine-grained venue discovery from multimodal data[J]. IEEE transactions on neural networks and learning systems. 2018;30(4):1250\u20138.","journal-title":"IEEE transactions on neural networks and learning systems."},{"key":"10287_CR226","doi-asserted-by":"crossref","unstructured":"Liu W, Qiu JL, Zheng WL, et al. Comparing recognition performance and robustness of multimodal deep learning models for multimodal emotion recognition[J]. IEEE Transactions on Cognitive and Developmental Systems. 2021;14(2):715\u201329.","DOI":"10.1109\/TCDS.2021.3071170"},{"issue":"17","key":"10287_CR227","doi-asserted-by":"publisher","first-page":"24477","DOI":"10.1007\/s11042-022-12435-9","volume":"81","author":"S Deshmukh","year":"2022","unstructured":"Deshmukh S, Abhyankar A, Kelkar S. DCCA and DMCCA framework for multimodal biometric system[J]. Multimed Tools Appl. 2022;81(17):24477\u201391.","journal-title":"Multimed Tools Appl."},{"key":"10287_CR228","unstructured":"Cevher D, Zepf S, Klinger R. Towards multimodal emotion recognition in German speech events in cars using transfer learning[J]. arXiv preprint arXiv:1909.02764, 2019."},{"key":"10287_CR229","doi-asserted-by":"crossref","unstructured":"Xi D, Zhou J, Xu W, et\u00a0al. Discrete emotion synchronicity and video engagement on social media: a moment-to-moment analysis[J]. Int J Electron Commerce. 2024:1-37.","DOI":"10.1080\/10864415.2023.2295072"},{"key":"10287_CR230","doi-asserted-by":"crossref","unstructured":"Lv Y, Liu Z, Li G. Context-aware interaction network for RGB-T semantic segmentation[J]. IEEE Trans Multimed. 2024.","DOI":"10.1109\/TMM.2023.3349072"},{"key":"10287_CR231","doi-asserted-by":"crossref","unstructured":"Ai W, Zhang F C, Meng T, et\u00a0al. A two-stage multimodal emotion recognition model based on graph contrastive learning[J]. arXiv preprint arXiv:2401.01495, 2024.","DOI":"10.1109\/ICPADS60453.2023.00067"},{"key":"10287_CR232","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2023.101587","volume":"85","author":"Y Wan","year":"2024","unstructured":"Wan Y, Chen Y, Lin J, et al. A knowledge-augmented heterogeneous graph convolutional network for aspect-level multimodal sentiment analysis[J]. Comput Speech Lang. 2024;85: 101587.","journal-title":"Comput Speech Lang."},{"key":"10287_CR233","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2023.102085","volume":"103","author":"P Tiwari","year":"2024","unstructured":"Tiwari P, Zhang L, Qu Z, et al. Quantum Fuzzy Neural Network for multimodal sentiment and sarcasm detection[J]. Inform Fusion. 2024;103: 102085.","journal-title":"Inform Fusion."},{"key":"10287_CR234","doi-asserted-by":"publisher","first-page":"110","DOI":"10.1016\/j.patrec.2023.11.029","volume":"177","author":"J Li","year":"2024","unstructured":"Li J, Li L, Sun R, et al. MMAN-M2: multiple multi-head attentions network based on encoder with missing modalities[J]. Pattern Recogn Lett. 2024;177:110\u201320.","journal-title":"Pattern Recogn Lett."},{"key":"10287_CR235","doi-asserted-by":"crossref","unstructured":"Zuo H, Liu R, Zhao J, et\u00a0al. Exploiting modality-invariant feature for robust multimodal emotion recognition with missing modalities[C]. In: ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 2023:1-5.","DOI":"10.1109\/ICASSP49357.2023.10095836"},{"key":"10287_CR236","doi-asserted-by":"crossref","unstructured":"Li M, Yang D, Zhang L. Towards robust multimodal sentiment analysis under uncertain signal missing[J]. IEEE Signal Process Lett. 2023.","DOI":"10.1109\/LSP.2023.3324552"},{"key":"10287_CR237","doi-asserted-by":"crossref","unstructured":"Mou L, Zhao Y, Zhou C, et\u00a0al. Driver emotion recognition with a hybrid attentional multimodal fusion framework[J]. IEEE Trans Affect Comput. 2023.","DOI":"10.1109\/TAFFC.2023.3250460"},{"key":"10287_CR238","doi-asserted-by":"publisher","DOI":"10.1016\/j.imavis.2022.104483","volume":"123","author":"A Kumar","year":"2022","unstructured":"Kumar A, Sharma K, Sharma A. MEmoR: a multimodal emotion recognition using affective biomarkers for smart prediction of emotional health for people analytics in smart industries[J]. Image Vis Comput. 2022;123: 104483.","journal-title":"Image Vis Comput."},{"key":"10287_CR239","doi-asserted-by":"crossref","unstructured":"Chong L, Jin M, He Y. EmoChat: bringing multimodal emotion detection to mobile conversation[C]. In: 2019 5th International Conference on Big Data Computing and Communications (BIGCOM). IEEE, 2019:213-221.","DOI":"10.1109\/BIGCOM.2019.00037"}],"container-title":["Cognitive Computation"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s12559-024-10287-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s12559-024-10287-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s12559-024-10287-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,20]],"date-time":"2024-11-20T22:55:43Z","timestamp":1732143343000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s12559-024-10287-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,6,1]]},"references-count":239,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2024,7]]}},"alternative-id":["10287"],"URL":"https:\/\/doi.org\/10.1007\/s12559-024-10287-z","relation":{},"ISSN":["1866-9956","1866-9964"],"issn-type":[{"value":"1866-9956","type":"print"},{"value":"1866-9964","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,6,1]]},"assertion":[{"value":"18 September 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 April 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"1 June 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"This article does not contain any studies with human participants or animals performed by any of the authors.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical Approval"}},{"value":"The authors declare no conflict of interest.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of Interest"}}]}}